{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XKPPNAT67HQAG2SJ7EO3J2DBUV","short_pith_number":"pith:XKPPNAT6","schema_version":"1.0","canonical_sha256":"ba9ef6827ef9e0036a49f91db4e861a55936872980251f55bd96c53bf0324d0c","source":{"kind":"arxiv","id":"2406.09455","version":1},"attestation_state":"computed","paper":{"title":"Pandora: Towards General World Model with Natural Language Actions and Video States","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Eric P. Xing, Guangyi Liu, Jiannan Xiang, Qiyue Gao, Shibo Hao, Tianhua Tao, Yemin Shi, Yi Gu, Yuheng Zha, Yuting Ning, Zeyu Feng, Zhengzhong Liu, Zhiting Hu","submitted_at":"2024-06-12T18:55:51Z","abstract_excerpt":"World models simulate future states of the world in response to different actions. They facilitate interactive content creation and provides a foundation for grounded, long-horizon reasoning. Current foundation models do not fully meet the capabilities of general world models: large language models (LLMs) are constrained by their reliance on language modality and their limited understanding of the physical world, while video models lack interactive action control over the world simulations. This paper makes a step towards building a general world model by introducing Pandora, a hybrid autoregr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.09455","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-06-12T18:55:51Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"64d3665304696b67eeedd38d420965d18fe1aff4e7c8dea929fc0155db0126cc","abstract_canon_sha256":"ac0d246b2ac89c1ac96115c26adb520ab3663d2b1dcddfb96827ac5404ead97d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:31:54.297280Z","signature_b64":"LY5hpK4+rL9JhDuNh3WeFQkABPSuconR4XYPy5gOf70fYphG9KljXSJdgxo+22t3EufjdfoQJiM3gD9K2S4mDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ba9ef6827ef9e0036a49f91db4e861a55936872980251f55bd96c53bf0324d0c","last_reissued_at":"2026-07-05T08:31:54.296852Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:31:54.296852Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Pandora: Towards General World Model with Natural Language Actions and Video States","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Eric P. Xing, Guangyi Liu, Jiannan Xiang, Qiyue Gao, Shibo Hao, Tianhua Tao, Yemin Shi, Yi Gu, Yuheng Zha, Yuting Ning, Zeyu Feng, Zhengzhong Liu, Zhiting Hu","submitted_at":"2024-06-12T18:55:51Z","abstract_excerpt":"World models simulate future states of the world in response to different actions. They facilitate interactive content creation and provides a foundation for grounded, long-horizon reasoning. Current foundation models do not fully meet the capabilities of general world models: large language models (LLMs) are constrained by their reliance on language modality and their limited understanding of the physical world, while video models lack interactive action control over the world simulations. This paper makes a step towards building a general world model by introducing Pandora, a hybrid autoregr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.09455","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.09455/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.09455","created_at":"2026-07-05T08:31:54.296918+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.09455v1","created_at":"2026-07-05T08:31:54.296918+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.09455","created_at":"2026-07-05T08:31:54.296918+00:00"},{"alias_kind":"pith_short_12","alias_value":"XKPPNAT67HQA","created_at":"2026-07-05T08:31:54.296918+00:00"},{"alias_kind":"pith_short_16","alias_value":"XKPPNAT67HQAG2SJ","created_at":"2026-07-05T08:31:54.296918+00:00"},{"alias_kind":"pith_short_8","alias_value":"XKPPNAT6","created_at":"2026-07-05T08:31:54.296918+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":15,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06559","citing_title":"RynnWorld-4D: 4D Embodied World Models for Robotic Manipulation","ref_index":28,"is_internal_anchor":true},{"citing_arxiv_id":"2606.27537","citing_title":"MemoBench: Benchmarking World Modeling in Dynamically Changing Environments","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09507","citing_title":"Prisma-World: Camera-Controllable Multi-Agent Video World Model","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27537","citing_title":"MemoBench: Benchmarking World Modeling in Dynamically Changing Environments","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27537","citing_title":"MemoBench: Benchmarking World Modeling in Dynamically Changing Environments","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27537","citing_title":"MemoBench: Benchmarking World Modeling in Dynamically Changing Environments","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00133","citing_title":"World Models: A Comprehensive Survey of Architectures, Methodologies, Reasoning Paradigms, and Applications","ref_index":160,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12191","citing_title":"Agentic Environment Engineering for Large Language Models: A Survey of Environment Modeling, Synthesis, Evaluation, and Application","ref_index":194,"is_internal_anchor":false},{"citing_arxiv_id":"2410.05363","citing_title":"Towards World Simulator: Crafting Physical Commonsense-Based Benchmark for Video Generation","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2503.22020","citing_title":"CoT-VLA: Visual Chain-of-Thought Reasoning for Vision-Language-Action Models","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2505.12705","citing_title":"DreamGen: Unlocking Generalization in Robot Learning through Video World Models","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08780","citing_title":"Toward Hardware-Agnostic Quadrupedal World Models via Morphology Conditioning","ref_index":75,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08719","citing_title":"LMGenDrive: Bridging Multimodal Understanding and Generative World Modeling for End-to-End Driving","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2503.14734","citing_title":"GR00T N1: An Open Foundation Model for Generalist Humanoid Robots","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17749","citing_title":"Ego-InBetween: Generating Object State Transitions in Ego-Centric Videos","ref_index":42,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XKPPNAT67HQAG2SJ7EO3J2DBUV","json":"https://pith.science/pith/XKPPNAT67HQAG2SJ7EO3J2DBUV.json","graph_json":"https://pith.science/api/pith-number/XKPPNAT67HQAG2SJ7EO3J2DBUV/graph.json","events_json":"https://pith.science/api/pith-number/XKPPNAT67HQAG2SJ7EO3J2DBUV/events.json","paper":"https://pith.science/paper/XKPPNAT6"},"agent_actions":{"view_html":"https://pith.science/pith/XKPPNAT67HQAG2SJ7EO3J2DBUV","download_json":"https://pith.science/pith/XKPPNAT67HQAG2SJ7EO3J2DBUV.json","view_paper":"https://pith.science/paper/XKPPNAT6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.09455&json=true","fetch_graph":"https://pith.science/api/pith-number/XKPPNAT67HQAG2SJ7EO3J2DBUV/graph.json","fetch_events":"https://pith.science/api/pith-number/XKPPNAT67HQAG2SJ7EO3J2DBUV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XKPPNAT67HQAG2SJ7EO3J2DBUV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XKPPNAT67HQAG2SJ7EO3J2DBUV/action/storage_attestation","attest_author":"https://pith.science/pith/XKPPNAT67HQAG2SJ7EO3J2DBUV/action/author_attestation","sign_citation":"https://pith.science/pith/XKPPNAT67HQAG2SJ7EO3J2DBUV/action/citation_signature","submit_replication":"https://pith.science/pith/XKPPNAT67HQAG2SJ7EO3J2DBUV/action/replication_record"}},"created_at":"2026-07-05T08:31:54.296918+00:00","updated_at":"2026-07-05T08:31:54.296918+00:00"}