{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:7HFMFSIRJSRZVBDU7L56VAE54E","short_pith_number":"pith:7HFMFSIR","schema_version":"1.0","canonical_sha256":"f9cac2c9114ca39a8474fafbea809de127716d437992198102e116a37bb3c309","source":{"kind":"arxiv","id":"2501.04003","version":1},"attestation_state":"computed","paper":{"title":"Are VLMs Ready for Autonomous Driving? An Empirical Study from the Reliability, Data, and Metric Perspectives","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.CV","authors_text":"Chonghao Sima, Liang Pan, Lingdong Kong, Qi Alfred Chen, Shaoyuan Xie, Wenwei Zhang, Yuhao Dong, Ziwei Liu","submitted_at":"2025-01-07T18:59:55Z","abstract_excerpt":"Recent advancements in Vision-Language Models (VLMs) have sparked interest in their use for autonomous driving, particularly in generating interpretable driving decisions through natural language. However, the assumption that VLMs inherently provide visually grounded, reliable, and interpretable explanations for driving remains largely unexamined. To address this gap, we introduce DriveBench, a benchmark dataset designed to evaluate VLM reliability across 17 settings (clean, corrupted, and text-only inputs), encompassing 19,200 frames, 20,498 question-answer pairs, three question types, four m"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.04003","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2025-01-07T18:59:55Z","cross_cats_sorted":["cs.RO"],"title_canon_sha256":"5792826c3958591b28d1bbf4a892aca73b3c56eeefae6c072e658e027dac7b29","abstract_canon_sha256":"f52930b2a87703bce186c5f7413d11789021e4f821038fabd0237e04dc4258e8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:58:14.234437Z","signature_b64":"/3dbR0Kuf9X262x0h5tR1RXXdPsG7+Ds9YxH1aJLk9xgHyfJ2zJyqQCU7KtgXyonpZjmgVNDeQLvsxcd2Gc2Ag==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f9cac2c9114ca39a8474fafbea809de127716d437992198102e116a37bb3c309","last_reissued_at":"2026-07-05T09:58:14.234005Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:58:14.234005Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Are VLMs Ready for Autonomous Driving? An Empirical Study from the Reliability, Data, and Metric Perspectives","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.CV","authors_text":"Chonghao Sima, Liang Pan, Lingdong Kong, Qi Alfred Chen, Shaoyuan Xie, Wenwei Zhang, Yuhao Dong, Ziwei Liu","submitted_at":"2025-01-07T18:59:55Z","abstract_excerpt":"Recent advancements in Vision-Language Models (VLMs) have sparked interest in their use for autonomous driving, particularly in generating interpretable driving decisions through natural language. However, the assumption that VLMs inherently provide visually grounded, reliable, and interpretable explanations for driving remains largely unexamined. To address this gap, we introduce DriveBench, a benchmark dataset designed to evaluate VLM reliability across 17 settings (clean, corrupted, and text-only inputs), encompassing 19,200 frames, 20,498 question-answer pairs, three question types, four m"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.04003","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.04003/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.04003","created_at":"2026-07-05T09:58:14.234060+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.04003v1","created_at":"2026-07-05T09:58:14.234060+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.04003","created_at":"2026-07-05T09:58:14.234060+00:00"},{"alias_kind":"pith_short_12","alias_value":"7HFMFSIRJSRZ","created_at":"2026-07-05T09:58:14.234060+00:00"},{"alias_kind":"pith_short_16","alias_value":"7HFMFSIRJSRZVBDU","created_at":"2026-07-05T09:58:14.234060+00:00"},{"alias_kind":"pith_short_8","alias_value":"7HFMFSIR","created_at":"2026-07-05T09:58:14.234060+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2607.05783","citing_title":"Benchmarking the Robustness of Autonomous Driving to Environmental Illusions: A Lane Perception Perspective","ref_index":104,"is_internal_anchor":true},{"citing_arxiv_id":"2604.18483","citing_title":"Steadily moving semi-infinite fracture in plane poroelasticity","ref_index":102,"is_internal_anchor":true},{"citing_arxiv_id":"2606.30220","citing_title":"From Accuracy to Visual Dependence: Auditing and Filtering Modality Collapse in Traffic VideoQA","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2602.18600","citing_title":"MapTab: A Diagnostic Benchmark for Long-Horizon Multi-Criteria Multimodal Reasoning on Heterogeneous Topological Graphs","ref_index":84,"is_internal_anchor":false},{"citing_arxiv_id":"2506.05442","citing_title":"Structured Labeling Enables Faster Vision-Language Models for End-to-End Autonomous Driving","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2511.00088","citing_title":"Alpamayo-R1: Bridging Reasoning and Action Prediction for Generalizable Autonomous Driving in the Long Tail","ref_index":101,"is_internal_anchor":false},{"citing_arxiv_id":"2511.13397","citing_title":"Descriptor: Distance-Annotated Traffic Perception Question Answering (DTPQA)","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2602.18600","citing_title":"MapTab: A Diagnostic Benchmark for Long-Horizon Multi-Criteria Multimodal Reasoning on Heterogeneous Topological Graphs","ref_index":84,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04857","citing_title":"The Blind Spot of Adaptation: Quantifying and Mitigating Forgetting in Fine-tuned Driving Models","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18484","citing_title":"XEmbodied: A Foundation Model with Enhanced Geometric and Physical Cues for Large-Scale Embodied Environments","ref_index":102,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00907","citing_title":"TRIP-Evaluate: An Open Multimodal Benchmark for Evaluating Large Models in Transportation","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7HFMFSIRJSRZVBDU7L56VAE54E","json":"https://pith.science/pith/7HFMFSIRJSRZVBDU7L56VAE54E.json","graph_json":"https://pith.science/api/pith-number/7HFMFSIRJSRZVBDU7L56VAE54E/graph.json","events_json":"https://pith.science/api/pith-number/7HFMFSIRJSRZVBDU7L56VAE54E/events.json","paper":"https://pith.science/paper/7HFMFSIR"},"agent_actions":{"view_html":"https://pith.science/pith/7HFMFSIRJSRZVBDU7L56VAE54E","download_json":"https://pith.science/pith/7HFMFSIRJSRZVBDU7L56VAE54E.json","view_paper":"https://pith.science/paper/7HFMFSIR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.04003&json=true","fetch_graph":"https://pith.science/api/pith-number/7HFMFSIRJSRZVBDU7L56VAE54E/graph.json","fetch_events":"https://pith.science/api/pith-number/7HFMFSIRJSRZVBDU7L56VAE54E/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7HFMFSIRJSRZVBDU7L56VAE54E/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7HFMFSIRJSRZVBDU7L56VAE54E/action/storage_attestation","attest_author":"https://pith.science/pith/7HFMFSIRJSRZVBDU7L56VAE54E/action/author_attestation","sign_citation":"https://pith.science/pith/7HFMFSIRJSRZVBDU7L56VAE54E/action/citation_signature","submit_replication":"https://pith.science/pith/7HFMFSIRJSRZVBDU7L56VAE54E/action/replication_record"}},"created_at":"2026-07-05T09:58:14.234060+00:00","updated_at":"2026-07-05T09:58:14.234060+00:00"}