{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ZHPTHM4ERJVFVZKNWU23MWA3QS","short_pith_number":"pith:ZHPTHM4E","schema_version":"1.0","canonical_sha256":"c9df33b3848a6a5ae54db535b6581b84a08049d295cc825ca4b03eebfa04cbf5","source":{"kind":"arxiv","id":"2507.06710","version":2},"attestation_state":"computed","paper":{"title":"Spatial-Temporal Aware Visuomotor Diffusion Policy Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Kuanning Wang, Longfei Liang, Xiangyang Xue, Yanwei Fu, Yikai Wang, Zhenyang Liu","submitted_at":"2025-07-09T10:08:15Z","abstract_excerpt":"Visual imitation learning is effective for robots to learn versatile tasks. However, many existing methods rely on behavior cloning with supervised historical trajectories, limiting their 3D spatial and 4D spatiotemporal awareness. Consequently, these methods struggle to capture the 3D structures and 4D spatiotemporal relationships necessary for real-world deployment. In this work, we propose 4D Diffusion Policy (DP4), a novel visual imitation learning method that incorporates spatiotemporal awareness into diffusion-based policies. Unlike traditional approaches that rely on trajectory cloning,"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.06710","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2025-07-09T10:08:15Z","cross_cats_sorted":[],"title_canon_sha256":"814a4543754d94714aa84a15f9c61057d641da804a4375e46545da84b676a0a2","abstract_canon_sha256":"d453b0bdeebdcfc5a4b749a961a746c8c1e1d453d4829ad2ed1e88d14613a3e6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:35:58.121205Z","signature_b64":"3w+mBQrwPTDJ+9OzZuTiw0sEtN/VdgavWow2V4f/Wo8RNJQ5yp9qKDPUzuN8BvEVCFKJxJkYC83R53gwD5LmBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c9df33b3848a6a5ae54db535b6581b84a08049d295cc825ca4b03eebfa04cbf5","last_reissued_at":"2026-07-05T11:35:58.120726Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:35:58.120726Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Spatial-Temporal Aware Visuomotor Diffusion Policy Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Kuanning Wang, Longfei Liang, Xiangyang Xue, Yanwei Fu, Yikai Wang, Zhenyang Liu","submitted_at":"2025-07-09T10:08:15Z","abstract_excerpt":"Visual imitation learning is effective for robots to learn versatile tasks. However, many existing methods rely on behavior cloning with supervised historical trajectories, limiting their 3D spatial and 4D spatiotemporal awareness. Consequently, these methods struggle to capture the 3D structures and 4D spatiotemporal relationships necessary for real-world deployment. In this work, we propose 4D Diffusion Policy (DP4), a novel visual imitation learning method that incorporates spatiotemporal awareness into diffusion-based policies. Unlike traditional approaches that rely on trajectory cloning,"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.06710","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.06710/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.06710","created_at":"2026-07-05T11:35:58.120790+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.06710v2","created_at":"2026-07-05T11:35:58.120790+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.06710","created_at":"2026-07-05T11:35:58.120790+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZHPTHM4ERJVF","created_at":"2026-07-05T11:35:58.120790+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZHPTHM4ERJVFVZKN","created_at":"2026-07-05T11:35:58.120790+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZHPTHM4E","created_at":"2026-07-05T11:35:58.120790+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.15938","citing_title":"VADF: Vision-Adaptive Diffusion Policy Framework for Efficient Robotic Manipulation","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZHPTHM4ERJVFVZKNWU23MWA3QS","json":"https://pith.science/pith/ZHPTHM4ERJVFVZKNWU23MWA3QS.json","graph_json":"https://pith.science/api/pith-number/ZHPTHM4ERJVFVZKNWU23MWA3QS/graph.json","events_json":"https://pith.science/api/pith-number/ZHPTHM4ERJVFVZKNWU23MWA3QS/events.json","paper":"https://pith.science/paper/ZHPTHM4E"},"agent_actions":{"view_html":"https://pith.science/pith/ZHPTHM4ERJVFVZKNWU23MWA3QS","download_json":"https://pith.science/pith/ZHPTHM4ERJVFVZKNWU23MWA3QS.json","view_paper":"https://pith.science/paper/ZHPTHM4E","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.06710&json=true","fetch_graph":"https://pith.science/api/pith-number/ZHPTHM4ERJVFVZKNWU23MWA3QS/graph.json","fetch_events":"https://pith.science/api/pith-number/ZHPTHM4ERJVFVZKNWU23MWA3QS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZHPTHM4ERJVFVZKNWU23MWA3QS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZHPTHM4ERJVFVZKNWU23MWA3QS/action/storage_attestation","attest_author":"https://pith.science/pith/ZHPTHM4ERJVFVZKNWU23MWA3QS/action/author_attestation","sign_citation":"https://pith.science/pith/ZHPTHM4ERJVFVZKNWU23MWA3QS/action/citation_signature","submit_replication":"https://pith.science/pith/ZHPTHM4ERJVFVZKNWU23MWA3QS/action/replication_record"}},"created_at":"2026-07-05T11:35:58.120790+00:00","updated_at":"2026-07-05T11:35:58.120790+00:00"}