{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:TYMIINE6LSIFIWE67CLIQNZWAW","short_pith_number":"pith:TYMIINE6","schema_version":"1.0","canonical_sha256":"9e1884349e5c9054589ef8968837360580a24cf85933fb8c53f31d638baeeb6e","source":{"kind":"arxiv","id":"2501.09757","version":1},"attestation_state":"computed","paper":{"title":"Distilling Multi-modal Large Language Models for Autonomous Driving","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.CV","authors_text":"Apratim Bhattacharyya, Deepti Hegde, Fatih Porikli, Hong Cai, Litian Liu, Rajeev Yasarla, Risheek Garrepalli, Shizhong Han, Shweta Mahajan, Vishal M. Patel","submitted_at":"2025-01-16T18:59:53Z","abstract_excerpt":"Autonomous driving demands safe motion planning, especially in critical \"long-tail\" scenarios. Recent end-to-end autonomous driving systems leverage large language models (LLMs) as planners to improve generalizability to rare events. However, using LLMs at test time introduces high computational costs. To address this, we propose DiMA, an end-to-end autonomous driving system that maintains the efficiency of an LLM-free (or vision-based) planner while leveraging the world knowledge of an LLM. DiMA distills the information from a multi-modal LLM to a vision-based end-to-end planner through a set"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.09757","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CV","submitted_at":"2025-01-16T18:59:53Z","cross_cats_sorted":["cs.RO"],"title_canon_sha256":"f2426ad2cbebd1e6cf76ca32e316f472fe73a4ab8574c98276d636e0813a776d","abstract_canon_sha256":"b2cec06f35b842ab618653306a741eeff75aaafe9dff094485d34ff6d1a20d6f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:01:58.197968Z","signature_b64":"3fuMjrYBvRqdqkCXWVyKWVs052I4vRLEZT4wBDjcop2x0Dhkq6ZI14CjLl+1gwQTgFGafxW8ZGjaxJRccuaMAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9e1884349e5c9054589ef8968837360580a24cf85933fb8c53f31d638baeeb6e","last_reissued_at":"2026-07-05T10:01:58.197485Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:01:58.197485Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Distilling Multi-modal Large Language Models for Autonomous Driving","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.CV","authors_text":"Apratim Bhattacharyya, Deepti Hegde, Fatih Porikli, Hong Cai, Litian Liu, Rajeev Yasarla, Risheek Garrepalli, Shizhong Han, Shweta Mahajan, Vishal M. Patel","submitted_at":"2025-01-16T18:59:53Z","abstract_excerpt":"Autonomous driving demands safe motion planning, especially in critical \"long-tail\" scenarios. Recent end-to-end autonomous driving systems leverage large language models (LLMs) as planners to improve generalizability to rare events. However, using LLMs at test time introduces high computational costs. To address this, we propose DiMA, an end-to-end autonomous driving system that maintains the efficiency of an LLM-free (or vision-based) planner while leveraging the world knowledge of an LLM. DiMA distills the information from a multi-modal LLM to a vision-based end-to-end planner through a set"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.09757","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.09757/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.09757","created_at":"2026-07-05T10:01:58.197547+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.09757v1","created_at":"2026-07-05T10:01:58.197547+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.09757","created_at":"2026-07-05T10:01:58.197547+00:00"},{"alias_kind":"pith_short_12","alias_value":"TYMIINE6LSIF","created_at":"2026-07-05T10:01:58.197547+00:00"},{"alias_kind":"pith_short_16","alias_value":"TYMIINE6LSIFIWE6","created_at":"2026-07-05T10:01:58.197547+00:00"},{"alias_kind":"pith_short_8","alias_value":"TYMIINE6","created_at":"2026-07-05T10:01:58.197547+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2506.13757","citing_title":"AutoVLA: A Vision-Language-Action Model for End-to-End Autonomous Driving with Adaptive Reasoning and Reinforcement Fine-Tuning","ref_index":77,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10564","citing_title":"DeepSight: Long-Horizon World Modeling via Latent States Prediction for End-to-End Autonomous Driving","ref_index":100,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TYMIINE6LSIFIWE67CLIQNZWAW","json":"https://pith.science/pith/TYMIINE6LSIFIWE67CLIQNZWAW.json","graph_json":"https://pith.science/api/pith-number/TYMIINE6LSIFIWE67CLIQNZWAW/graph.json","events_json":"https://pith.science/api/pith-number/TYMIINE6LSIFIWE67CLIQNZWAW/events.json","paper":"https://pith.science/paper/TYMIINE6"},"agent_actions":{"view_html":"https://pith.science/pith/TYMIINE6LSIFIWE67CLIQNZWAW","download_json":"https://pith.science/pith/TYMIINE6LSIFIWE67CLIQNZWAW.json","view_paper":"https://pith.science/paper/TYMIINE6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.09757&json=true","fetch_graph":"https://pith.science/api/pith-number/TYMIINE6LSIFIWE67CLIQNZWAW/graph.json","fetch_events":"https://pith.science/api/pith-number/TYMIINE6LSIFIWE67CLIQNZWAW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TYMIINE6LSIFIWE67CLIQNZWAW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TYMIINE6LSIFIWE67CLIQNZWAW/action/storage_attestation","attest_author":"https://pith.science/pith/TYMIINE6LSIFIWE67CLIQNZWAW/action/author_attestation","sign_citation":"https://pith.science/pith/TYMIINE6LSIFIWE67CLIQNZWAW/action/citation_signature","submit_replication":"https://pith.science/pith/TYMIINE6LSIFIWE67CLIQNZWAW/action/replication_record"}},"created_at":"2026-07-05T10:01:58.197547+00:00","updated_at":"2026-07-05T10:01:58.197547+00:00"}