{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:LVMCN3262VDIW7RY2JJLAAWYZN","short_pith_number":"pith:LVMCN326","schema_version":"1.0","canonical_sha256":"5d5826ef5ed5468b7e38d252b002d8cb6bc703dfdddd1d59756a529e63c76bca","source":{"kind":"arxiv","id":"2507.17744","version":1},"attestation_state":"computed","paper":{"title":"Yume: An Interactive World Generation Model","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.HC"],"primary_cat":"cs.CV","authors_text":"Chuanhao Li, Jiangmiao Pang, Kaipeng Zhang, Mingmin Chi, Shaoheng Lin, Tong He, Wenshuo Peng, Xiaofeng Mao, Yu Qiao, Zhen Li","submitted_at":"2025-07-23T17:57:09Z","abstract_excerpt":"Yume aims to use images, text, or videos to create an interactive, realistic, and dynamic world, which allows exploration and control using peripheral devices or neural signals. In this report, we present a preview version of \\method, which creates a dynamic world from an input image and allows exploration of the world using keyboard actions. To achieve this high-fidelity and interactive video world generation, we introduce a well-designed framework, which consists of four main components, including camera motion quantization, video generation architecture, advanced sampler, and model accelera"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.17744","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-07-23T17:57:09Z","cross_cats_sorted":["cs.AI","cs.HC"],"title_canon_sha256":"cb2049bef257a433a4b398ca7fe12a433f5b1f952044e249f5c9600f60e377c8","abstract_canon_sha256":"acddc04c954ab447a19f8adb69cd47f44e774549d9944149eeffaaeeb4f7e4e3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:42:17.669375Z","signature_b64":"kC7vqf0JlI3gIy5Wm51rZ6Et86dhr+Kg0E/HRTAw8FaKLdyzYiHZOIXk/artOGBkXyaRjfvMVP2Z740O7FP1Bw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5d5826ef5ed5468b7e38d252b002d8cb6bc703dfdddd1d59756a529e63c76bca","last_reissued_at":"2026-07-05T11:42:17.668929Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:42:17.668929Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Yume: An Interactive World Generation Model","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.HC"],"primary_cat":"cs.CV","authors_text":"Chuanhao Li, Jiangmiao Pang, Kaipeng Zhang, Mingmin Chi, Shaoheng Lin, Tong He, Wenshuo Peng, Xiaofeng Mao, Yu Qiao, Zhen Li","submitted_at":"2025-07-23T17:57:09Z","abstract_excerpt":"Yume aims to use images, text, or videos to create an interactive, realistic, and dynamic world, which allows exploration and control using peripheral devices or neural signals. In this report, we present a preview version of \\method, which creates a dynamic world from an input image and allows exploration of the world using keyboard actions. To achieve this high-fidelity and interactive video world generation, we introduce a well-designed framework, which consists of four main components, including camera motion quantization, video generation architecture, advanced sampler, and model accelera"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.17744","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.17744/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.17744","created_at":"2026-07-05T11:42:17.668984+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.17744v1","created_at":"2026-07-05T11:42:17.668984+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.17744","created_at":"2026-07-05T11:42:17.668984+00:00"},{"alias_kind":"pith_short_12","alias_value":"LVMCN3262VDI","created_at":"2026-07-05T11:42:17.668984+00:00"},{"alias_kind":"pith_short_16","alias_value":"LVMCN3262VDIW7RY","created_at":"2026-07-05T11:42:17.668984+00:00"},{"alias_kind":"pith_short_8","alias_value":"LVMCN326","created_at":"2026-07-05T11:42:17.668984+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":29,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06291","citing_title":"AlayaWorld: Long-Horizon and Playable Video World Generation","ref_index":45,"is_internal_anchor":true},{"citing_arxiv_id":"2606.26694","citing_title":"PhysEditWorld: A Large-Scale Dataset Toward Physics-Editable World Models","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18180","citing_title":"EgoCS-400K: An Egocentric Gameplay Dataset for World Models","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.16449","citing_title":"PermaVid: Consistent Video Generation Across Edits via Disentangled Context Memory","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02517","citing_title":"WorldDirector: Building Controllable World Simulators with Persistent Dynamic Memory","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2606.13376","citing_title":"MoVerse: Real-Time Video World Modeling with Panoramic Gaussian Scaffold","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11129","citing_title":"WorldOlympiad: Can Your World Model Survive a Triathlon?","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09803","citing_title":"Echo-Memory: A Controlled Study of Memory in Action World Models","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09507","citing_title":"Prisma-World: Camera-Controllable Multi-Agent Video World Model","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07967","citing_title":"DisCo: World Models with Discrete Camera Motion Control","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06147","citing_title":"WorldFly: A World-Model-Based Vision-Language-Action Model for UAV Navigation","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01060","citing_title":"RoboWorld: Fast and Reliable Neural Simulators for Generalist Robot Policy Evaluation","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02575","citing_title":"From Zero to Hero: Training-Free Custom Concept Spawning in World Models","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02436","citing_title":"Geometry-Aware Implicit Memory for Video World Models","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01164","citing_title":"Towards Interactive Video World Modeling: Frontiers, Challenges, Benchmarks, and Future Trends","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26694","citing_title":"PhysEditWorld: A Large-Scale Dataset Toward Physics-Editable World Models","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25077","citing_title":"WorldCraft: From Camera Navigation to Object Manipulation in Interactive Video World Models","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25874","citing_title":"WBench: A Comprehensive Multi-turn Benchmark for Interactive Video World Model Evaluation","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00793","citing_title":"MBench: A Comprehensive Benchmark on Memory Capability for Video World Models","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20961","citing_title":"Preserve, Reveal, Expand: Faithful 4D Video Editing with Region-Aware Conditioning","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19957","citing_title":"World-Ego Modeling for Long-Horizon Evolution in Hybrid Embodied Tasks","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2508.13009","citing_title":"Matrix-game 2.0: An open-source real-time and streaming interactive world model","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2512.14614","citing_title":"WorldPlay: Towards Long-Term Geometric Consistency for Real-Time Interactive World Modeling","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2509.22622","citing_title":"LongLive: Real-time Interactive Long Video Generation","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13111","citing_title":"Pyramid Forcing: Head-Aware Pyramid KV Cache Policy for High-Quality Long Video Generation","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LVMCN3262VDIW7RY2JJLAAWYZN","json":"https://pith.science/pith/LVMCN3262VDIW7RY2JJLAAWYZN.json","graph_json":"https://pith.science/api/pith-number/LVMCN3262VDIW7RY2JJLAAWYZN/graph.json","events_json":"https://pith.science/api/pith-number/LVMCN3262VDIW7RY2JJLAAWYZN/events.json","paper":"https://pith.science/paper/LVMCN326"},"agent_actions":{"view_html":"https://pith.science/pith/LVMCN3262VDIW7RY2JJLAAWYZN","download_json":"https://pith.science/pith/LVMCN3262VDIW7RY2JJLAAWYZN.json","view_paper":"https://pith.science/paper/LVMCN326","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.17744&json=true","fetch_graph":"https://pith.science/api/pith-number/LVMCN3262VDIW7RY2JJLAAWYZN/graph.json","fetch_events":"https://pith.science/api/pith-number/LVMCN3262VDIW7RY2JJLAAWYZN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LVMCN3262VDIW7RY2JJLAAWYZN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LVMCN3262VDIW7RY2JJLAAWYZN/action/storage_attestation","attest_author":"https://pith.science/pith/LVMCN3262VDIW7RY2JJLAAWYZN/action/author_attestation","sign_citation":"https://pith.science/pith/LVMCN3262VDIW7RY2JJLAAWYZN/action/citation_signature","submit_replication":"https://pith.science/pith/LVMCN3262VDIW7RY2JJLAAWYZN/action/replication_record"}},"created_at":"2026-07-05T11:42:17.668984+00:00","updated_at":"2026-07-05T11:42:17.668984+00:00"}