{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:HRBRMMZWYOPE4OSV6WBGZU3P3K","short_pith_number":"pith:HRBRMMZW","schema_version":"1.0","canonical_sha256":"3c43163336c39e4e3a55f5826cd36fdab2e06d7d229032369ea07212397e9ea2","source":{"kind":"arxiv","id":"2409.03272","version":1},"attestation_state":"computed","paper":{"title":"OccLLaMA: An Occupancy-Language-Action Generative World Model for Autonomous Driving","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.CV","authors_text":"Julong Wei, Pengfei Li, Qingda Hu, Shanshuai Yuan, Wenchao Ding, Zhongxue Gan","submitted_at":"2024-09-05T06:30:01Z","abstract_excerpt":"The rise of multi-modal large language models(MLLMs) has spurred their applications in autonomous driving. Recent MLLM-based methods perform action by learning a direct mapping from perception to action, neglecting the dynamics of the world and the relations between action and world dynamics. In contrast, human beings possess world model that enables them to simulate the future states based on 3D internal visual representation and plan actions accordingly. To this end, we propose OccLLaMA, an occupancy-language-action generative world model, which uses semantic occupancy as a general visual re"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.03272","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-09-05T06:30:01Z","cross_cats_sorted":["cs.RO"],"title_canon_sha256":"840aa6cf5633dd99edc4d9aa5b687010c2f8b766442f1c0cd946c8c17d363e6c","abstract_canon_sha256":"2952fddfdffb9670adf80dd0cdbcec6ba0f23909f2d83a93df3d2a4e88fde80d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:03:28.909120Z","signature_b64":"klTWXSX1hnkILjE0m27U0xm49InF4ZBNKstdiPHjcfL//aVXKOlF7rfJM0BChZeloz9hD7tudSzbgTSx3ll2AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3c43163336c39e4e3a55f5826cd36fdab2e06d7d229032369ea07212397e9ea2","last_reissued_at":"2026-07-05T09:03:28.908590Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:03:28.908590Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"OccLLaMA: An Occupancy-Language-Action Generative World Model for Autonomous Driving","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.CV","authors_text":"Julong Wei, Pengfei Li, Qingda Hu, Shanshuai Yuan, Wenchao Ding, Zhongxue Gan","submitted_at":"2024-09-05T06:30:01Z","abstract_excerpt":"The rise of multi-modal large language models(MLLMs) has spurred their applications in autonomous driving. Recent MLLM-based methods perform action by learning a direct mapping from perception to action, neglecting the dynamics of the world and the relations between action and world dynamics. In contrast, human beings possess world model that enables them to simulate the future states based on 3D internal visual representation and plan actions accordingly. To this end, we propose OccLLaMA, an occupancy-language-action generative world model, which uses semantic occupancy as a general visual re"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.03272","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.03272/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.03272","created_at":"2026-07-05T09:03:28.908659+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.03272v1","created_at":"2026-07-05T09:03:28.908659+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.03272","created_at":"2026-07-05T09:03:28.908659+00:00"},{"alias_kind":"pith_short_12","alias_value":"HRBRMMZWYOPE","created_at":"2026-07-05T09:03:28.908659+00:00"},{"alias_kind":"pith_short_16","alias_value":"HRBRMMZWYOPE4OSV","created_at":"2026-07-05T09:03:28.908659+00:00"},{"alias_kind":"pith_short_8","alias_value":"HRBRMMZW","created_at":"2026-07-05T09:03:28.908659+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.05645","citing_title":"Discrete-WAM: Unified Discrete Vision-Action Token Editing for World-Policy Learning","ref_index":78,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27644","citing_title":"CascadeOcc: Rethinking 3D Occupancy World Models with Cascaded VQ Representations","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26113","citing_title":"AnyScene: Towards Highly Controllable Driving Scene Generation at Anywhere and Beyond","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27038","citing_title":"TPS-Drive: Task-Guided Representation Purification for VLM-based Autonomous Driving","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2602.22667","citing_title":"Monocular Open Vocabulary Occupancy Prediction for Indoor Scenes","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17682","citing_title":"GEM: Gaussian Evolution Model for Occupancy Forecasting and Motion Planning","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2511.22039","citing_title":"SparseWorld-TC: Trajectory-Conditioned Sparse Occupancy World Model","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2603.27507","citing_title":"Chat-Scene++: Exploiting Context-Rich Object Identification for 3D LLM","ref_index":74,"is_internal_anchor":false},{"citing_arxiv_id":"2604.28196","citing_title":"HERMES++: Toward a Unified Driving World Model for 3D Scene Understanding and Generation","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09701","citing_title":"DriveFuture: Future-Aware Latent World Models for Autonomous Driving","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12857","citing_title":"Artificial Intelligence for Modeling and Simulation of Mixed Automated and Human Traffic","ref_index":153,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HRBRMMZWYOPE4OSV6WBGZU3P3K","json":"https://pith.science/pith/HRBRMMZWYOPE4OSV6WBGZU3P3K.json","graph_json":"https://pith.science/api/pith-number/HRBRMMZWYOPE4OSV6WBGZU3P3K/graph.json","events_json":"https://pith.science/api/pith-number/HRBRMMZWYOPE4OSV6WBGZU3P3K/events.json","paper":"https://pith.science/paper/HRBRMMZW"},"agent_actions":{"view_html":"https://pith.science/pith/HRBRMMZWYOPE4OSV6WBGZU3P3K","download_json":"https://pith.science/pith/HRBRMMZWYOPE4OSV6WBGZU3P3K.json","view_paper":"https://pith.science/paper/HRBRMMZW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.03272&json=true","fetch_graph":"https://pith.science/api/pith-number/HRBRMMZWYOPE4OSV6WBGZU3P3K/graph.json","fetch_events":"https://pith.science/api/pith-number/HRBRMMZWYOPE4OSV6WBGZU3P3K/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HRBRMMZWYOPE4OSV6WBGZU3P3K/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HRBRMMZWYOPE4OSV6WBGZU3P3K/action/storage_attestation","attest_author":"https://pith.science/pith/HRBRMMZWYOPE4OSV6WBGZU3P3K/action/author_attestation","sign_citation":"https://pith.science/pith/HRBRMMZWYOPE4OSV6WBGZU3P3K/action/citation_signature","submit_replication":"https://pith.science/pith/HRBRMMZWYOPE4OSV6WBGZU3P3K/action/replication_record"}},"created_at":"2026-07-05T09:03:28.908659+00:00","updated_at":"2026-07-05T09:03:28.908659+00:00"}