{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:O4WWUKROAVMEV7IBNGGHR7JH46","short_pith_number":"pith:O4WWUKRO","schema_version":"1.0","canonical_sha256":"772d6a2a2e05584afd01698c78fd27e79cf116c94a02bd7e77f05a8ea1dd29df","source":{"kind":"arxiv","id":"2311.13549","version":1},"attestation_state":"computed","paper":{"title":"ADriver-I: A General World Model for Autonomous Driving","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.CV","authors_text":"Chi Zhang, Fan Jia, Tiancai Wang, Weixin Mao, Xiangyu Zhang, Yingfei Liu, Yucheng Zhao, Yuqing Wen","submitted_at":"2023-11-22T17:44:29Z","abstract_excerpt":"Typically, autonomous driving adopts a modular design, which divides the full stack into perception, prediction, planning and control parts. Though interpretable, such modular design tends to introduce a substantial amount of redundancy. Recently, multimodal large language models (MLLM) and diffusion techniques have demonstrated their superior performance on comprehension and generation ability. In this paper, we first introduce the concept of interleaved vision-action pair, which unifies the format of visual features and control signals. Based on the vision-action pairs, we construct a genera"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.13549","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-11-22T17:44:29Z","cross_cats_sorted":["cs.RO"],"title_canon_sha256":"db10ee5ce3b543a4a506d6c12c600535739eee300433a0a84f747c7d994c35e5","abstract_canon_sha256":"0f7191311d7441451a752bc31acba92e807a9db05b63484f20a8397029520852"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:15:39.988645Z","signature_b64":"XjK6isDvARXsJDyQAPyWoZxnVUkF8iMX2neMArpubIOQErFEj2KPugVmUOCM8H/vE0XdL1GUcBSSX+DyG2CEAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"772d6a2a2e05584afd01698c78fd27e79cf116c94a02bd7e77f05a8ea1dd29df","last_reissued_at":"2026-07-05T07:15:39.988111Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:15:39.988111Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ADriver-I: A General World Model for Autonomous Driving","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.CV","authors_text":"Chi Zhang, Fan Jia, Tiancai Wang, Weixin Mao, Xiangyu Zhang, Yingfei Liu, Yucheng Zhao, Yuqing Wen","submitted_at":"2023-11-22T17:44:29Z","abstract_excerpt":"Typically, autonomous driving adopts a modular design, which divides the full stack into perception, prediction, planning and control parts. Though interpretable, such modular design tends to introduce a substantial amount of redundancy. Recently, multimodal large language models (MLLM) and diffusion techniques have demonstrated their superior performance on comprehension and generation ability. In this paper, we first introduce the concept of interleaved vision-action pair, which unifies the format of visual features and control signals. Based on the vision-action pairs, we construct a genera"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.13549","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.13549/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.13549","created_at":"2026-07-05T07:15:39.988186+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.13549v1","created_at":"2026-07-05T07:15:39.988186+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.13549","created_at":"2026-07-05T07:15:39.988186+00:00"},{"alias_kind":"pith_short_12","alias_value":"O4WWUKROAVME","created_at":"2026-07-05T07:15:39.988186+00:00"},{"alias_kind":"pith_short_16","alias_value":"O4WWUKROAVMEV7IB","created_at":"2026-07-05T07:15:39.988186+00:00"},{"alias_kind":"pith_short_8","alias_value":"O4WWUKRO","created_at":"2026-07-05T07:15:39.988186+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.31226","citing_title":"ForgeDrive: Bidirectional Cross-Conditioning for Unified Visual-Action Generation in Autonomous Driving","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24354","citing_title":"SparseWorld: Enhancing End-to-End Autonomous Driving via World Models with Sparse Scene Representation","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2504.18576","citing_title":"DriVerse: Navigation World Model for Driving Simulation via Multimodal Trajectory Prompting and Motion Alignment","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2411.02385","citing_title":"How Far is Video Generation from World Model: A Physical Law Perspective","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2406.08481","citing_title":"Enhancing End-to-End Autonomous Driving with Latent World Model","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2603.28489","citing_title":"Video Generation Models as World Models: Efficient Paradigms, Architectures and Algorithms","ref_index":188,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11550","citing_title":"The DAWN of World-Action Interactive Models","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2604.28196","citing_title":"HERMES++: Toward a Unified Driving World Model for 3D Scene Understanding and Generation","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09701","citing_title":"DriveFuture: Future-Aware Latent World Models for Autonomous Driving","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25329","citing_title":"ProDrive: Proactive Planning for Autonomous Driving via Ego-Environment Co-Evolution","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09059","citing_title":"Learning Vision-Language-Action World Models for Autonomous Driving","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2501.03575","citing_title":"Cosmos World Foundation Model Platform for Physical AI","ref_index":86,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/O4WWUKROAVMEV7IBNGGHR7JH46","json":"https://pith.science/pith/O4WWUKROAVMEV7IBNGGHR7JH46.json","graph_json":"https://pith.science/api/pith-number/O4WWUKROAVMEV7IBNGGHR7JH46/graph.json","events_json":"https://pith.science/api/pith-number/O4WWUKROAVMEV7IBNGGHR7JH46/events.json","paper":"https://pith.science/paper/O4WWUKRO"},"agent_actions":{"view_html":"https://pith.science/pith/O4WWUKROAVMEV7IBNGGHR7JH46","download_json":"https://pith.science/pith/O4WWUKROAVMEV7IBNGGHR7JH46.json","view_paper":"https://pith.science/paper/O4WWUKRO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.13549&json=true","fetch_graph":"https://pith.science/api/pith-number/O4WWUKROAVMEV7IBNGGHR7JH46/graph.json","fetch_events":"https://pith.science/api/pith-number/O4WWUKROAVMEV7IBNGGHR7JH46/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/O4WWUKROAVMEV7IBNGGHR7JH46/action/timestamp_anchor","attest_storage":"https://pith.science/pith/O4WWUKROAVMEV7IBNGGHR7JH46/action/storage_attestation","attest_author":"https://pith.science/pith/O4WWUKROAVMEV7IBNGGHR7JH46/action/author_attestation","sign_citation":"https://pith.science/pith/O4WWUKROAVMEV7IBNGGHR7JH46/action/citation_signature","submit_replication":"https://pith.science/pith/O4WWUKROAVMEV7IBNGGHR7JH46/action/replication_record"}},"created_at":"2026-07-05T07:15:39.988186+00:00","updated_at":"2026-07-05T07:15:39.988186+00:00"}