{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XESNOLBI45SFVIKFTWJX47HFOF","short_pith_number":"pith:XESNOLBI","schema_version":"1.0","canonical_sha256":"b924d72c28e7645aa1459d937e7ce5717cb72ea57e15c04d9f7348158e89de80","source":{"kind":"arxiv","id":"2412.19505","version":2},"attestation_state":"computed","paper":{"title":"DrivingWorld: Constructing World Model for Autonomous Driving via Video GPT","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Junyuan Deng, Mingkai Jia, Ping Tan, Qian Zhang, Wei Yin, Xiaotao Hu, Xiaoxiao Long, Xiaoyang Guo","submitted_at":"2024-12-27T07:44:07Z","abstract_excerpt":"Recent successes in autoregressive (AR) generation models, such as the GPT series in natural language processing, have motivated efforts to replicate this success in visual tasks. Some works attempt to extend this approach to autonomous driving by building video-based world models capable of generating realistic future video sequences and predicting ego states. However, prior works tend to produce unsatisfactory results, as the classic GPT framework is designed to handle 1D contextual information, such as text, and lacks the inherent ability to model the spatial and temporal dynamics essential"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.19505","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-12-27T07:44:07Z","cross_cats_sorted":[],"title_canon_sha256":"0876424590fd3b118f8ec97dc63e93ac482b8990fc6a75192a4f004de2b4503c","abstract_canon_sha256":"2cc0e57d3010cd6c29a3b7a1b5945e4415446eae78d5d026fd118ba44a56b1b6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:55:16.418602Z","signature_b64":"yOhEd0DSwIKO8CygUbvlhkDe0DgfoZJbMMgBxmloTRx3hkfqJ9ggLAEJyvRXTkoVPFYTcUpDnxalCIYO/Y2qAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b924d72c28e7645aa1459d937e7ce5717cb72ea57e15c04d9f7348158e89de80","last_reissued_at":"2026-07-05T09:55:16.418160Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:55:16.418160Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DrivingWorld: Constructing World Model for Autonomous Driving via Video GPT","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Junyuan Deng, Mingkai Jia, Ping Tan, Qian Zhang, Wei Yin, Xiaotao Hu, Xiaoxiao Long, Xiaoyang Guo","submitted_at":"2024-12-27T07:44:07Z","abstract_excerpt":"Recent successes in autoregressive (AR) generation models, such as the GPT series in natural language processing, have motivated efforts to replicate this success in visual tasks. Some works attempt to extend this approach to autonomous driving by building video-based world models capable of generating realistic future video sequences and predicting ego states. However, prior works tend to produce unsatisfactory results, as the classic GPT framework is designed to handle 1D contextual information, such as text, and lacks the inherent ability to model the spatial and temporal dynamics essential"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.19505","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.19505/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.19505","created_at":"2026-07-05T09:55:16.418213+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.19505v2","created_at":"2026-07-05T09:55:16.418213+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.19505","created_at":"2026-07-05T09:55:16.418213+00:00"},{"alias_kind":"pith_short_12","alias_value":"XESNOLBI45SF","created_at":"2026-07-05T09:55:16.418213+00:00"},{"alias_kind":"pith_short_16","alias_value":"XESNOLBI45SFVIKF","created_at":"2026-07-05T09:55:16.418213+00:00"},{"alias_kind":"pith_short_8","alias_value":"XESNOLBI","created_at":"2026-07-05T09:55:16.418213+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.05645","citing_title":"Discrete-WAM: Unified Discrete Vision-Action Token Editing for World-Policy Learning","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01164","citing_title":"Towards Interactive Video World Modeling: Frontiers, Challenges, Benchmarks, and Future Trends","ref_index":213,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26113","citing_title":"AnyScene: Towards Highly Controllable Driving Scene Generation at Anywhere and Beyond","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2504.18576","citing_title":"DriVerse: Navigation World Model for Driving Simulation via Multimodal Trajectory Prompting and Motion Alignment","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2512.23421","citing_title":"DriveLaW:Unifying Planning and Video Generation in a Latent Driving World","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10426","citing_title":"CoWorld-VLA: Thinking in a Multi-Expert World Model for Autonomous Driving","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2604.28196","citing_title":"HERMES++: Toward a Unified Driving World Model for 3D Scene Understanding and Generation","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10426","citing_title":"CoWorld-VLA: Thinking in a Multi-Expert World Model for Autonomous Driving","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12857","citing_title":"Artificial Intelligence for Modeling and Simulation of Mixed Automated and Human Traffic","ref_index":145,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09059","citing_title":"Learning Vision-Language-Action World Models for Autonomous Driving","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XESNOLBI45SFVIKFTWJX47HFOF","json":"https://pith.science/pith/XESNOLBI45SFVIKFTWJX47HFOF.json","graph_json":"https://pith.science/api/pith-number/XESNOLBI45SFVIKFTWJX47HFOF/graph.json","events_json":"https://pith.science/api/pith-number/XESNOLBI45SFVIKFTWJX47HFOF/events.json","paper":"https://pith.science/paper/XESNOLBI"},"agent_actions":{"view_html":"https://pith.science/pith/XESNOLBI45SFVIKFTWJX47HFOF","download_json":"https://pith.science/pith/XESNOLBI45SFVIKFTWJX47HFOF.json","view_paper":"https://pith.science/paper/XESNOLBI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.19505&json=true","fetch_graph":"https://pith.science/api/pith-number/XESNOLBI45SFVIKFTWJX47HFOF/graph.json","fetch_events":"https://pith.science/api/pith-number/XESNOLBI45SFVIKFTWJX47HFOF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XESNOLBI45SFVIKFTWJX47HFOF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XESNOLBI45SFVIKFTWJX47HFOF/action/storage_attestation","attest_author":"https://pith.science/pith/XESNOLBI45SFVIKFTWJX47HFOF/action/author_attestation","sign_citation":"https://pith.science/pith/XESNOLBI45SFVIKFTWJX47HFOF/action/citation_signature","submit_replication":"https://pith.science/pith/XESNOLBI45SFVIKFTWJX47HFOF/action/replication_record"}},"created_at":"2026-07-05T09:55:16.418213+00:00","updated_at":"2026-07-05T09:55:16.418213+00:00"}