{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:IB3TXDV4G3FIOPLFMZPEKQBOVN","short_pith_number":"pith:IB3TXDV4","schema_version":"1.0","canonical_sha256":"40773b8ebc36ca873d65665e45402eab688e050a6ae56c48ed62343fc65b69a1","source":{"kind":"arxiv","id":"2507.21045","version":2},"attestation_state":"computed","paper":{"title":"Reconstructing 4D Spatial Intelligence: A Survey","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chengfeng Zhao, Fangzhou Hong, Jiahao Lu, Wenping Wang, Xin Li, Yuan Liu, Yukang Cao, Zhaoxi Chen, Zhisheng Huang, Zhuowen Shen, Ziwei Liu","submitted_at":"2025-07-28T17:59:02Z","abstract_excerpt":"Reconstructing 4D spatial intelligence from visual observations has long been a central yet challenging task in computer vision, with broad real-world applications. These range from entertainment domains like movies, where the focus is often on reconstructing fundamental visual elements, to embodied AI, which emphasizes interaction modeling and physical realism. Fueled by rapid advances in 3D representations and deep learning architectures, the field has evolved quickly, outpacing the scope of previous surveys. Additionally, existing surveys rarely offer a comprehensive analysis of the hierarc"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.21045","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"cs.CV","submitted_at":"2025-07-28T17:59:02Z","cross_cats_sorted":[],"title_canon_sha256":"5fb121b6cf7329d542e4967281c202135cb51fbb4fb26845432e21f56dc09ef3","abstract_canon_sha256":"723f56f42768ecbdc7fa00517420e513abe3e7a93a4450a21b84e142a2e05c33"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:47:27.166688Z","signature_b64":"2YuZbn0KthMUG3nC52s1mW7/kIoWibye13HC9utETUD3mOjV9WHkSEElLG0tO2pqZPFo0ERiXERse1PMMA8GCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"40773b8ebc36ca873d65665e45402eab688e050a6ae56c48ed62343fc65b69a1","last_reissued_at":"2026-07-05T11:47:27.166199Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:47:27.166199Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Reconstructing 4D Spatial Intelligence: A Survey","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chengfeng Zhao, Fangzhou Hong, Jiahao Lu, Wenping Wang, Xin Li, Yuan Liu, Yukang Cao, Zhaoxi Chen, Zhisheng Huang, Zhuowen Shen, Ziwei Liu","submitted_at":"2025-07-28T17:59:02Z","abstract_excerpt":"Reconstructing 4D spatial intelligence from visual observations has long been a central yet challenging task in computer vision, with broad real-world applications. These range from entertainment domains like movies, where the focus is often on reconstructing fundamental visual elements, to embodied AI, which emphasizes interaction modeling and physical realism. Fueled by rapid advances in 3D representations and deep learning architectures, the field has evolved quickly, outpacing the scope of previous surveys. Additionally, existing surveys rarely offer a comprehensive analysis of the hierarc"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.21045","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.21045/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.21045","created_at":"2026-07-05T11:47:27.166257+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.21045v2","created_at":"2026-07-05T11:47:27.166257+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.21045","created_at":"2026-07-05T11:47:27.166257+00:00"},{"alias_kind":"pith_short_12","alias_value":"IB3TXDV4G3FI","created_at":"2026-07-05T11:47:27.166257+00:00"},{"alias_kind":"pith_short_16","alias_value":"IB3TXDV4G3FIOPLF","created_at":"2026-07-05T11:47:27.166257+00:00"},{"alias_kind":"pith_short_8","alias_value":"IB3TXDV4","created_at":"2026-07-05T11:47:27.166257+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.31388","citing_title":"One Video, One World: Turning Monocular Video into Physical 4D Scenes","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2510.17568","citing_title":"PAGE-4D: Disentangled pose and geometry estimation for vggt-4d perception","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2601.10632","citing_title":"CoMoVi: Co-Generation of 3D Human Motions and Realistic Videos","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14462","citing_title":"Real2Sim in HOI: Toward Physically Plausible HOI Reconstruction from Monocular Videos","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07923","citing_title":"Stitch4D: Sparse Multi-Location 4D Urban Reconstruction via Spatio-Temporal Interpolation","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IB3TXDV4G3FIOPLFMZPEKQBOVN","json":"https://pith.science/pith/IB3TXDV4G3FIOPLFMZPEKQBOVN.json","graph_json":"https://pith.science/api/pith-number/IB3TXDV4G3FIOPLFMZPEKQBOVN/graph.json","events_json":"https://pith.science/api/pith-number/IB3TXDV4G3FIOPLFMZPEKQBOVN/events.json","paper":"https://pith.science/paper/IB3TXDV4"},"agent_actions":{"view_html":"https://pith.science/pith/IB3TXDV4G3FIOPLFMZPEKQBOVN","download_json":"https://pith.science/pith/IB3TXDV4G3FIOPLFMZPEKQBOVN.json","view_paper":"https://pith.science/paper/IB3TXDV4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.21045&json=true","fetch_graph":"https://pith.science/api/pith-number/IB3TXDV4G3FIOPLFMZPEKQBOVN/graph.json","fetch_events":"https://pith.science/api/pith-number/IB3TXDV4G3FIOPLFMZPEKQBOVN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IB3TXDV4G3FIOPLFMZPEKQBOVN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IB3TXDV4G3FIOPLFMZPEKQBOVN/action/storage_attestation","attest_author":"https://pith.science/pith/IB3TXDV4G3FIOPLFMZPEKQBOVN/action/author_attestation","sign_citation":"https://pith.science/pith/IB3TXDV4G3FIOPLFMZPEKQBOVN/action/citation_signature","submit_replication":"https://pith.science/pith/IB3TXDV4G3FIOPLFMZPEKQBOVN/action/replication_record"}},"created_at":"2026-07-05T11:47:27.166257+00:00","updated_at":"2026-07-05T11:47:27.166257+00:00"}