{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:EEEHFAGTZ6HOB7UFB4BSCJQOJE","short_pith_number":"pith:EEEHFAGT","schema_version":"1.0","canonical_sha256":"21087280d3cf8ee0fe850f0321260e493949f3cf85154cd826c7a10f04a21047","source":{"kind":"arxiv","id":"2504.20995","version":1},"attestation_state":"computed","paper":{"title":"TesserAct: Learning 4D Embodied World Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.CV","authors_text":"Chuang Gan, Haoyu Zhen, Hongxin Zhang, Junyan Li, Qiao Sun, Siyuan Zhou, Yilun Du","submitted_at":"2025-04-29T17:59:30Z","abstract_excerpt":"This paper presents an effective approach for learning novel 4D embodied world models, which predict the dynamic evolution of 3D scenes over time in response to an embodied agent's actions, providing both spatial and temporal consistency. We propose to learn a 4D world model by training on RGB-DN (RGB, Depth, and Normal) videos. This not only surpasses traditional 2D models by incorporating detailed shape, configuration, and temporal changes into their predictions, but also allows us to effectively learn accurate inverse dynamic models for an embodied agent. Specifically, we first extend exist"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.20995","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-04-29T17:59:30Z","cross_cats_sorted":["cs.RO"],"title_canon_sha256":"30d94ce6162d74810a844c4ac0ba0f12d84974dc37bde39d83d6d6b71e3bb52e","abstract_canon_sha256":"81ca0c6cc4b3f1147cd11750310797d305e286cdb82e91877c9af78312cf23f0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:55:52.688631Z","signature_b64":"QeSjMDZDdxIRsVuuD06Lkl22JU1J6zObeZscMrnnHrUORecsM42bzG9LRX/haAeWh8pry/ZDxuJoXOS3EubrAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"21087280d3cf8ee0fe850f0321260e493949f3cf85154cd826c7a10f04a21047","last_reissued_at":"2026-07-05T10:55:52.688138Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:55:52.688138Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TesserAct: Learning 4D Embodied World Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.CV","authors_text":"Chuang Gan, Haoyu Zhen, Hongxin Zhang, Junyan Li, Qiao Sun, Siyuan Zhou, Yilun Du","submitted_at":"2025-04-29T17:59:30Z","abstract_excerpt":"This paper presents an effective approach for learning novel 4D embodied world models, which predict the dynamic evolution of 3D scenes over time in response to an embodied agent's actions, providing both spatial and temporal consistency. We propose to learn a 4D world model by training on RGB-DN (RGB, Depth, and Normal) videos. This not only surpasses traditional 2D models by incorporating detailed shape, configuration, and temporal changes into their predictions, but also allows us to effectively learn accurate inverse dynamic models for an embodied agent. Specifically, we first extend exist"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.20995","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.20995/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.20995","created_at":"2026-07-05T10:55:52.688195+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.20995v1","created_at":"2026-07-05T10:55:52.688195+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.20995","created_at":"2026-07-05T10:55:52.688195+00:00"},{"alias_kind":"pith_short_12","alias_value":"EEEHFAGTZ6HO","created_at":"2026-07-05T10:55:52.688195+00:00"},{"alias_kind":"pith_short_16","alias_value":"EEEHFAGTZ6HOB7UF","created_at":"2026-07-05T10:55:52.688195+00:00"},{"alias_kind":"pith_short_8","alias_value":"EEEHFAGT","created_at":"2026-07-05T10:55:52.688195+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":37,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06216","citing_title":"MoWorld: A Flash World Model","ref_index":30,"is_internal_anchor":true},{"citing_arxiv_id":"2606.20781","citing_title":"World Action Models: A Survey","ref_index":202,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19531","citing_title":"ImageWAM: Do World Action Models Really Need Video Generation, or Just Image Editing?","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12403","citing_title":"World Pilot: Steering Vision-Language-Action Models with World-Action Priors","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01896","citing_title":"Divide and Conquer: Decoupled Representation Alignment for Multimodal World Models","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00148","citing_title":"3D Point World Models: Point Completion Enables More Accurate Dynamics Learning","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04264","citing_title":"UniCanvas: A Diffusion-base Unified Model for Text-in-Image Joint Generation","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01164","citing_title":"Towards Interactive Video World Modeling: Frontiers, Challenges, Benchmarks, and Future Trends","ref_index":189,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31388","citing_title":"One Video, One World: Turning Monocular Video into Physical 4D Scenes","ref_index":120,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01799","citing_title":"Embody4D: A Generalist Data Engine for Embodied 4D World Modeling","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23993","citing_title":"Nano World Models: A Minimalist Implementation of Future Video Prediction","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22882","citing_title":"GEM-4D: Geometry-Enhanced Video World Models for Robot Manipulation","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28128","citing_title":"PhysisForcing: Physics Reinforced World Simulator for Robotic Manipulation","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27589","citing_title":"What-If World: A Causal Benchmark for General World Models in Embodied Scenarios","ref_index":77,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00113","citing_title":"World Models for Robotic Manipulation: A Survey","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00133","citing_title":"World Models: A Comprehensive Survey of Architectures, Methodologies, Reasoning Paradigms, and Applications","ref_index":188,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03943","citing_title":"PointAction: 3D Points as Universal Action Representations for Robot Control","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22882","citing_title":"GEM-4D: Geometry-Enhanced Video World Models for Robot Manipulation","ref_index":70,"is_internal_anchor":false},{"citing_arxiv_id":"2507.01099","citing_title":"Geometry-aware 4D Video Generation for Robot Manipulation","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2505.21996","citing_title":"VRAG: Learning World Models for Interactive Video Generation","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2507.07982","citing_title":"Geometry Forcing: Marrying Video Diffusion and 3D Representation for Consistent World Modeling","ref_index":95,"is_internal_anchor":false},{"citing_arxiv_id":"2512.01773","citing_title":"IGen: Scalable Data Generation for Robot Learning from Open-World Images","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2512.15840","citing_title":"Large Video Planner Enables Generalizable Robot Control","ref_index":89,"is_internal_anchor":false},{"citing_arxiv_id":"2603.12639","citing_title":"RoboStereo: Dual-Tower 4D Embodied World Models for Unified Policy Optimization","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12090","citing_title":"World Action Models: The Next Frontier in Embodied AI","ref_index":72,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EEEHFAGTZ6HOB7UFB4BSCJQOJE","json":"https://pith.science/pith/EEEHFAGTZ6HOB7UFB4BSCJQOJE.json","graph_json":"https://pith.science/api/pith-number/EEEHFAGTZ6HOB7UFB4BSCJQOJE/graph.json","events_json":"https://pith.science/api/pith-number/EEEHFAGTZ6HOB7UFB4BSCJQOJE/events.json","paper":"https://pith.science/paper/EEEHFAGT"},"agent_actions":{"view_html":"https://pith.science/pith/EEEHFAGTZ6HOB7UFB4BSCJQOJE","download_json":"https://pith.science/pith/EEEHFAGTZ6HOB7UFB4BSCJQOJE.json","view_paper":"https://pith.science/paper/EEEHFAGT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.20995&json=true","fetch_graph":"https://pith.science/api/pith-number/EEEHFAGTZ6HOB7UFB4BSCJQOJE/graph.json","fetch_events":"https://pith.science/api/pith-number/EEEHFAGTZ6HOB7UFB4BSCJQOJE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EEEHFAGTZ6HOB7UFB4BSCJQOJE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EEEHFAGTZ6HOB7UFB4BSCJQOJE/action/storage_attestation","attest_author":"https://pith.science/pith/EEEHFAGTZ6HOB7UFB4BSCJQOJE/action/author_attestation","sign_citation":"https://pith.science/pith/EEEHFAGTZ6HOB7UFB4BSCJQOJE/action/citation_signature","submit_replication":"https://pith.science/pith/EEEHFAGTZ6HOB7UFB4BSCJQOJE/action/replication_record"}},"created_at":"2026-07-05T10:55:52.688195+00:00","updated_at":"2026-07-05T10:55:52.688195+00:00"}