{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:ZIPLG4A4ZS2MVP2ZGVSYMQSWYE","short_pith_number":"pith:ZIPLG4A4","schema_version":"1.0","canonical_sha256":"ca1eb3701cccb4cabf593565864256c125a5c7d86e87ad56a23724ec6b336b9c","source":{"kind":"arxiv","id":"2312.14150","version":3},"attestation_state":"computed","paper":{"title":"DriveLM: Driving with Graph Visual Question Answering","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Andreas Geiger, Chengen Xie, Chonghao Sima, Hanxue Zhang, Hongyang Li, Jens Bei{\\ss}wenger, Kashyap Chitta, Katrin Renz, Li Chen, Ping Luo","submitted_at":"2023-12-21T18:59:12Z","abstract_excerpt":"We study how vision-language models (VLMs) trained on web-scale data can be integrated into end-to-end driving systems to boost generalization and enable interactivity with human users. While recent approaches adapt VLMs to driving via single-round visual question answering (VQA), human drivers reason about decisions in multiple steps. Starting from the localization of key objects, humans estimate object interactions before taking actions. The key insight is that with our proposed task, Graph VQA, where we model graph-structured reasoning through perception, prediction and planning question-an"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.14150","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-12-21T18:59:12Z","cross_cats_sorted":[],"title_canon_sha256":"f7136dfd58072342df1d126f6eda36f55cb6be94eeb566ac21a3b9a5aec455bf","abstract_canon_sha256":"d39d361117d2aeb300921edb8c2f2cdbfaa645ea509163aac3e033078e37e27d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:01:31.777363Z","signature_b64":"FuStkSEuOa2afvLUNWNAjz8MkIJX2LBJvJpvIaM+rmVpdb0CGhSsUvlsJd9RxGbCyUTKddR1asjGlaPF4USkAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ca1eb3701cccb4cabf593565864256c125a5c7d86e87ad56a23724ec6b336b9c","last_reissued_at":"2026-07-05T10:01:31.776823Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:01:31.776823Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DriveLM: Driving with Graph Visual Question Answering","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Andreas Geiger, Chengen Xie, Chonghao Sima, Hanxue Zhang, Hongyang Li, Jens Bei{\\ss}wenger, Kashyap Chitta, Katrin Renz, Li Chen, Ping Luo","submitted_at":"2023-12-21T18:59:12Z","abstract_excerpt":"We study how vision-language models (VLMs) trained on web-scale data can be integrated into end-to-end driving systems to boost generalization and enable interactivity with human users. While recent approaches adapt VLMs to driving via single-round visual question answering (VQA), human drivers reason about decisions in multiple steps. Starting from the localization of key objects, humans estimate object interactions before taking actions. The key insight is that with our proposed task, Graph VQA, where we model graph-structured reasoning through perception, prediction and planning question-an"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.14150","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.14150/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.14150","created_at":"2026-07-05T10:01:31.776888+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.14150v3","created_at":"2026-07-05T10:01:31.776888+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.14150","created_at":"2026-07-05T10:01:31.776888+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZIPLG4A4ZS2M","created_at":"2026-07-05T10:01:31.776888+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZIPLG4A4ZS2MVP2Z","created_at":"2026-07-05T10:01:31.776888+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZIPLG4A4","created_at":"2026-07-05T10:01:31.776888+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":18,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21165","citing_title":"OmniV2X: A Generative Foundation Planner for Efficient End-to-End Cooperative Driving","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27365","citing_title":"LocateAnything: Fast and High-Quality Vision-Language Grounding with Parallel Box Decoding","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23270","citing_title":"ChainFlow-VLA: Causal Flow Planning with Vision-Language Models","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2411.18275","citing_title":"Visual Adversarial Attack on Vision-Language Models for Autonomous Driving","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2505.16278","citing_title":"DriveMoE: Mixture-of-Experts for Vision-Language-Action Model in End-to-End Autonomous Driving","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2506.05442","citing_title":"Structured Labeling Enables Faster Vision-Language Models for End-to-End Autonomous Driving","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2508.05269","citing_title":"B4DL: A Benchmark for 4D LiDAR LLM in Spatio-Temporal Understanding","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2512.14044","citing_title":"OmniDrive-R1: Reinforcement-driven Interleaved Multi-modal Chain-of-Thought for Trustworthy Vision-Language Autonomous Driving","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2503.07608","citing_title":"AlphaDrive: Unleashing the Power of VLMs in Autonomous Driving via Reinforcement Learning and Reasoning","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2408.13257","citing_title":"MME-RealWorld: Could Your Multimodal LLM Challenge High-Resolution Real-World Scenarios that are Difficult for Humans?","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2410.22313","citing_title":"Senna: Bridging Large Vision-Language Models and End-to-End Autonomous Driving","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14696","citing_title":"EponaV2: Driving World Model with Comprehensive Future Reasoning","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2402.12289","citing_title":"DriveVLM: The Convergence of Autonomous Driving and Large Vision-Language Models","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20460","citing_title":"CCTVBench: Contrastive Consistency Traffic VideoQA Benchmark for Multimodal LLMs","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17915","citing_title":"OneDrive: Unified Multi-Paradigm Driving with Vision-Language-Action Models","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05767","citing_title":"Beyond the Beep: Scalable Collision Anticipation and Real-Time Explainability with BADAS-2.0","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22851","citing_title":"EgoDyn-Bench: Evaluating Ego-Motion Understanding in Vision-Centric Foundation Models for Autonomous Driving","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00907","citing_title":"TRIP-Evaluate: An Open Multimodal Benchmark for Evaluating Large Models in Transportation","ref_index":25,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZIPLG4A4ZS2MVP2ZGVSYMQSWYE","json":"https://pith.science/pith/ZIPLG4A4ZS2MVP2ZGVSYMQSWYE.json","graph_json":"https://pith.science/api/pith-number/ZIPLG4A4ZS2MVP2ZGVSYMQSWYE/graph.json","events_json":"https://pith.science/api/pith-number/ZIPLG4A4ZS2MVP2ZGVSYMQSWYE/events.json","paper":"https://pith.science/paper/ZIPLG4A4"},"agent_actions":{"view_html":"https://pith.science/pith/ZIPLG4A4ZS2MVP2ZGVSYMQSWYE","download_json":"https://pith.science/pith/ZIPLG4A4ZS2MVP2ZGVSYMQSWYE.json","view_paper":"https://pith.science/paper/ZIPLG4A4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.14150&json=true","fetch_graph":"https://pith.science/api/pith-number/ZIPLG4A4ZS2MVP2ZGVSYMQSWYE/graph.json","fetch_events":"https://pith.science/api/pith-number/ZIPLG4A4ZS2MVP2ZGVSYMQSWYE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZIPLG4A4ZS2MVP2ZGVSYMQSWYE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZIPLG4A4ZS2MVP2ZGVSYMQSWYE/action/storage_attestation","attest_author":"https://pith.science/pith/ZIPLG4A4ZS2MVP2ZGVSYMQSWYE/action/author_attestation","sign_citation":"https://pith.science/pith/ZIPLG4A4ZS2MVP2ZGVSYMQSWYE/action/citation_signature","submit_replication":"https://pith.science/pith/ZIPLG4A4ZS2MVP2ZGVSYMQSWYE/action/replication_record"}},"created_at":"2026-07-05T10:01:31.776888+00:00","updated_at":"2026-07-05T10:01:31.776888+00:00"}