{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:B4NV64ENUUGIXK5MZAQUYENDQL","short_pith_number":"pith:B4NV64EN","schema_version":"1.0","canonical_sha256":"0f1b5f708da50c8babacc8214c11a382f1c3a7fa2123b1846be6989704a502b1","source":{"kind":"arxiv","id":"2504.14526","version":1},"attestation_state":"computed","paper":{"title":"Are Vision LLMs Road-Ready? A Comprehensive Benchmark for Safety-Critical Driving Video Understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Dawei Zhou, Feng Guo, Liang Shi, Longfeng Wu, Tong Zeng","submitted_at":"2025-04-20T07:50:44Z","abstract_excerpt":"Vision Large Language Models (VLLMs) have demonstrated impressive capabilities in general visual tasks such as image captioning and visual question answering. However, their effectiveness in specialized, safety-critical domains like autonomous driving remains largely unexplored. Autonomous driving systems require sophisticated scene understanding in complex environments, yet existing multimodal benchmarks primarily focus on normal driving conditions, failing to adequately assess VLLMs' performance in safety-critical scenarios. To address this, we introduce DVBench, a pioneering benchmark desig"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.14526","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-04-20T07:50:44Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"8343c948d13ba285c7ec5905178260a834c502a5b153ff0caf3e72665f92251e","abstract_canon_sha256":"fe0eb44e05adc48062f938d901a4b846893010356000b0b90ea179072b809e6a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:51:44.835260Z","signature_b64":"BGufCZzZvt7VLRl7cyaWoOzQFBLaB7Xt4yzX6+fCA5zRlSe7UW8V2mht0gpVQ0MfhNa+6vadaGxb54N0YMMeBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0f1b5f708da50c8babacc8214c11a382f1c3a7fa2123b1846be6989704a502b1","last_reissued_at":"2026-07-05T10:51:44.834713Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:51:44.834713Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Are Vision LLMs Road-Ready? A Comprehensive Benchmark for Safety-Critical Driving Video Understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Dawei Zhou, Feng Guo, Liang Shi, Longfeng Wu, Tong Zeng","submitted_at":"2025-04-20T07:50:44Z","abstract_excerpt":"Vision Large Language Models (VLLMs) have demonstrated impressive capabilities in general visual tasks such as image captioning and visual question answering. However, their effectiveness in specialized, safety-critical domains like autonomous driving remains largely unexplored. Autonomous driving systems require sophisticated scene understanding in complex environments, yet existing multimodal benchmarks primarily focus on normal driving conditions, failing to adequately assess VLLMs' performance in safety-critical scenarios. To address this, we introduce DVBench, a pioneering benchmark desig"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.14526","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.14526/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.14526","created_at":"2026-07-05T10:51:44.834770+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.14526v1","created_at":"2026-07-05T10:51:44.834770+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.14526","created_at":"2026-07-05T10:51:44.834770+00:00"},{"alias_kind":"pith_short_12","alias_value":"B4NV64ENUUGI","created_at":"2026-07-05T10:51:44.834770+00:00"},{"alias_kind":"pith_short_16","alias_value":"B4NV64ENUUGIXK5M","created_at":"2026-07-05T10:51:44.834770+00:00"},{"alias_kind":"pith_short_8","alias_value":"B4NV64EN","created_at":"2026-07-05T10:51:44.834770+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2509.25944","citing_title":"NuRisk: A Visual Question Answering Dataset for Agent-Level Risk Assessment in Autonomous Driving","ref_index":27,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/B4NV64ENUUGIXK5MZAQUYENDQL","json":"https://pith.science/pith/B4NV64ENUUGIXK5MZAQUYENDQL.json","graph_json":"https://pith.science/api/pith-number/B4NV64ENUUGIXK5MZAQUYENDQL/graph.json","events_json":"https://pith.science/api/pith-number/B4NV64ENUUGIXK5MZAQUYENDQL/events.json","paper":"https://pith.science/paper/B4NV64EN"},"agent_actions":{"view_html":"https://pith.science/pith/B4NV64ENUUGIXK5MZAQUYENDQL","download_json":"https://pith.science/pith/B4NV64ENUUGIXK5MZAQUYENDQL.json","view_paper":"https://pith.science/paper/B4NV64EN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.14526&json=true","fetch_graph":"https://pith.science/api/pith-number/B4NV64ENUUGIXK5MZAQUYENDQL/graph.json","fetch_events":"https://pith.science/api/pith-number/B4NV64ENUUGIXK5MZAQUYENDQL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/B4NV64ENUUGIXK5MZAQUYENDQL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/B4NV64ENUUGIXK5MZAQUYENDQL/action/storage_attestation","attest_author":"https://pith.science/pith/B4NV64ENUUGIXK5MZAQUYENDQL/action/author_attestation","sign_citation":"https://pith.science/pith/B4NV64ENUUGIXK5MZAQUYENDQL/action/citation_signature","submit_replication":"https://pith.science/pith/B4NV64ENUUGIXK5MZAQUYENDQL/action/replication_record"}},"created_at":"2026-07-05T10:51:44.834770+00:00","updated_at":"2026-07-05T10:51:44.834770+00:00"}