{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:3CJDADMCG6YG5P2OZCZQ6TPG7Q","short_pith_number":"pith:3CJDADMC","schema_version":"1.0","canonical_sha256":"d892300d8237b06ebf4ec8b30f4de6fc235d911a5b1bf44d728ba29a2316e43a","source":{"kind":"arxiv","id":"2505.11528","version":6},"attestation_state":"computed","paper":{"title":"LaDi-WM: A Latent Diffusion-based World Model for Predictive Manipulation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.RO","authors_text":"Jiazhao Zhang, Kai Xu, Ruizhen Hu, Shilong Zou, Xinwang Liu, Yuhang Huang","submitted_at":"2025-05-13T04:42:14Z","abstract_excerpt":"Predictive manipulation has recently gained considerable attention in the Embodied AI community due to its potential to improve robot policy performance by leveraging predicted states. However, generating accurate future visual states of robot-object interactions from world models remains a well-known challenge, particularly in achieving high-quality pixel-level representations. To this end, we propose LaDi-WM, a world model that predicts the latent space of future states using diffusion modeling. Specifically, LaDi-WM leverages the well-established latent space aligned with pre-trained Visual"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.11528","kind":"arxiv","version":6},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2025-05-13T04:42:14Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"aabb0f863456f74206cc2ab1156d19145094d6417af02987183ab681b918d22f","abstract_canon_sha256":"5a9d0e8aa37061cd795e571a120c509f8db9e36bbba7e56be23eef272e268fec"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:10:57.854890Z","signature_b64":"yO93lCjXqkb/Te5VvWBiXaWh/9hjXp7Tu+0vLnaGLg9YI0A7ZgPnfMLijvhlYkJWnRUwYNKhhdTDROTsMG6NAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d892300d8237b06ebf4ec8b30f4de6fc235d911a5b1bf44d728ba29a2316e43a","last_reissued_at":"2026-07-05T12:10:57.854296Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:10:57.854296Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LaDi-WM: A Latent Diffusion-based World Model for Predictive Manipulation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.RO","authors_text":"Jiazhao Zhang, Kai Xu, Ruizhen Hu, Shilong Zou, Xinwang Liu, Yuhang Huang","submitted_at":"2025-05-13T04:42:14Z","abstract_excerpt":"Predictive manipulation has recently gained considerable attention in the Embodied AI community due to its potential to improve robot policy performance by leveraging predicted states. However, generating accurate future visual states of robot-object interactions from world models remains a well-known challenge, particularly in achieving high-quality pixel-level representations. To this end, we propose LaDi-WM, a world model that predicts the latent space of future states using diffusion modeling. Specifically, LaDi-WM leverages the well-established latent space aligned with pre-trained Visual"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.11528","kind":"arxiv","version":6},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.11528/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.11528","created_at":"2026-07-05T12:10:57.854367+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.11528v6","created_at":"2026-07-05T12:10:57.854367+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.11528","created_at":"2026-07-05T12:10:57.854367+00:00"},{"alias_kind":"pith_short_12","alias_value":"3CJDADMCG6YG","created_at":"2026-07-05T12:10:57.854367+00:00"},{"alias_kind":"pith_short_16","alias_value":"3CJDADMCG6YG5P2O","created_at":"2026-07-05T12:10:57.854367+00:00"},{"alias_kind":"pith_short_8","alias_value":"3CJDADMC","created_at":"2026-07-05T12:10:57.854367+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25295","citing_title":"DynaMOMA: Instantaneous Prediction of Grasp Poses for Mobile Manipulation of Dynamic Objects","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26916","citing_title":"PhysRAG: Enhancing Physics-Awareness in Video Generation via Retrieval-Augmented Generation","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22966","citing_title":"Attacking the Trusted Imagination: Oracle-Level Integrity Attacks on Imagine-then-Act World Models","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18375","citing_title":"PAIWorld: A 3D-Consistent World Foundation Model for Robotic Manipulation","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01955","citing_title":"WALL-WM: Carving World Action Modeling at the Event Joints","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2512.21714","citing_title":"AstraNav-World: World Model for Foresight Control and Consistency","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2507.04447","citing_title":"DreamVLA: A Vision-Language-Action Model Dreamed with Comprehensive World Knowledge","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24622","citing_title":"CF-VLA: Efficient Coarse-to-Fine Action Generation for Vision-Language-Action Policies","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16592","citing_title":"Human Cognition in Machines: A Unified Perspective of World Models","ref_index":72,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3CJDADMCG6YG5P2OZCZQ6TPG7Q","json":"https://pith.science/pith/3CJDADMCG6YG5P2OZCZQ6TPG7Q.json","graph_json":"https://pith.science/api/pith-number/3CJDADMCG6YG5P2OZCZQ6TPG7Q/graph.json","events_json":"https://pith.science/api/pith-number/3CJDADMCG6YG5P2OZCZQ6TPG7Q/events.json","paper":"https://pith.science/paper/3CJDADMC"},"agent_actions":{"view_html":"https://pith.science/pith/3CJDADMCG6YG5P2OZCZQ6TPG7Q","download_json":"https://pith.science/pith/3CJDADMCG6YG5P2OZCZQ6TPG7Q.json","view_paper":"https://pith.science/paper/3CJDADMC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.11528&json=true","fetch_graph":"https://pith.science/api/pith-number/3CJDADMCG6YG5P2OZCZQ6TPG7Q/graph.json","fetch_events":"https://pith.science/api/pith-number/3CJDADMCG6YG5P2OZCZQ6TPG7Q/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3CJDADMCG6YG5P2OZCZQ6TPG7Q/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3CJDADMCG6YG5P2OZCZQ6TPG7Q/action/storage_attestation","attest_author":"https://pith.science/pith/3CJDADMCG6YG5P2OZCZQ6TPG7Q/action/author_attestation","sign_citation":"https://pith.science/pith/3CJDADMCG6YG5P2OZCZQ6TPG7Q/action/citation_signature","submit_replication":"https://pith.science/pith/3CJDADMCG6YG5P2OZCZQ6TPG7Q/action/replication_record"}},"created_at":"2026-07-05T12:10:57.854367+00:00","updated_at":"2026-07-05T12:10:57.854367+00:00"}