{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:PLG27OOU3UVG4IRSLFODCMOSNW","short_pith_number":"pith:PLG27OOU","schema_version":"1.0","canonical_sha256":"7acdafb9d4dd2a6e2232595c3131d26d8d47e761d7eb58680ac3db8b760df3c4","source":{"kind":"arxiv","id":"2407.07860","version":2},"attestation_state":"computed","paper":{"title":"Controlling Space and Time with Diffusion Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Andrea Tagliasacchi, Daniel Watson, David J. Fleet, Lala Li, Saurabh Saxena","submitted_at":"2024-07-10T17:23:33Z","abstract_excerpt":"We present 4DiM, a cascaded diffusion model for 4D novel view synthesis (NVS), supporting generation with arbitrary camera trajectories and timestamps, in natural scenes, conditioned on one or more images. With a novel architecture and sampling procedure, we enable training on a mixture of 3D (with camera pose), 4D (pose+time) and video (time but no pose) data, which greatly improves generalization to unseen images and camera pose trajectories over prior works that focus on limited domains (e.g., object centric). 4DiM is the first-ever NVS method with intuitive metric-scale camera pose control"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.07860","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-07-10T17:23:33Z","cross_cats_sorted":[],"title_canon_sha256":"8eb26324de9443e9ec128b7087516f73cdec7ad5149cf8d0aab884088fb50e13","abstract_canon_sha256":"a47f7d46cb0454ae2d4bb55c3927b1f3675a3961bf41484ab319862f14d35803"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:51:23.835736Z","signature_b64":"E9w89iqTltjKRxfkN2i2bm+cEm5p5jQ1VhWt0qkbsrcjsZdsawSlMj5OqxxyVmrwen2uVIqEeO09BT2rlQsNBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7acdafb9d4dd2a6e2232595c3131d26d8d47e761d7eb58680ac3db8b760df3c4","last_reissued_at":"2026-07-05T10:51:23.835235Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:51:23.835235Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Controlling Space and Time with Diffusion Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Andrea Tagliasacchi, Daniel Watson, David J. Fleet, Lala Li, Saurabh Saxena","submitted_at":"2024-07-10T17:23:33Z","abstract_excerpt":"We present 4DiM, a cascaded diffusion model for 4D novel view synthesis (NVS), supporting generation with arbitrary camera trajectories and timestamps, in natural scenes, conditioned on one or more images. With a novel architecture and sampling procedure, we enable training on a mixture of 3D (with camera pose), 4D (pose+time) and video (time but no pose) data, which greatly improves generalization to unseen images and camera pose trajectories over prior works that focus on limited domains (e.g., object centric). 4DiM is the first-ever NVS method with intuitive metric-scale camera pose control"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.07860","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.07860/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.07860","created_at":"2026-07-05T10:51:23.835294+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.07860v2","created_at":"2026-07-05T10:51:23.835294+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.07860","created_at":"2026-07-05T10:51:23.835294+00:00"},{"alias_kind":"pith_short_12","alias_value":"PLG27OOU3UVG","created_at":"2026-07-05T10:51:23.835294+00:00"},{"alias_kind":"pith_short_16","alias_value":"PLG27OOU3UVG4IRS","created_at":"2026-07-05T10:51:23.835294+00:00"},{"alias_kind":"pith_short_8","alias_value":"PLG27OOU","created_at":"2026-07-05T10:51:23.835294+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.01164","citing_title":"Towards Interactive Video World Modeling: Frontiers, Challenges, Benchmarks, and Future Trends","ref_index":116,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31535","citing_title":"RayDer: Scalable Self-Supervised Novel View Synthesis from Real-World Video","ref_index":78,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22131","citing_title":"Feed-forward Motion In-betweening for Any 4D","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2411.13549","citing_title":"KFC-W: Generating 3D-Consistent Videos from Unposed Internet Photos","ref_index":75,"is_internal_anchor":false},{"citing_arxiv_id":"2602.04876","citing_title":"PerpetualWonder: Long-Horizon Action-Conditioned 4D Scene Generation","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2511.00062","citing_title":"World Simulation with Video Foundation Models for Physical AI","ref_index":84,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PLG27OOU3UVG4IRSLFODCMOSNW","json":"https://pith.science/pith/PLG27OOU3UVG4IRSLFODCMOSNW.json","graph_json":"https://pith.science/api/pith-number/PLG27OOU3UVG4IRSLFODCMOSNW/graph.json","events_json":"https://pith.science/api/pith-number/PLG27OOU3UVG4IRSLFODCMOSNW/events.json","paper":"https://pith.science/paper/PLG27OOU"},"agent_actions":{"view_html":"https://pith.science/pith/PLG27OOU3UVG4IRSLFODCMOSNW","download_json":"https://pith.science/pith/PLG27OOU3UVG4IRSLFODCMOSNW.json","view_paper":"https://pith.science/paper/PLG27OOU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.07860&json=true","fetch_graph":"https://pith.science/api/pith-number/PLG27OOU3UVG4IRSLFODCMOSNW/graph.json","fetch_events":"https://pith.science/api/pith-number/PLG27OOU3UVG4IRSLFODCMOSNW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PLG27OOU3UVG4IRSLFODCMOSNW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PLG27OOU3UVG4IRSLFODCMOSNW/action/storage_attestation","attest_author":"https://pith.science/pith/PLG27OOU3UVG4IRSLFODCMOSNW/action/author_attestation","sign_citation":"https://pith.science/pith/PLG27OOU3UVG4IRSLFODCMOSNW/action/citation_signature","submit_replication":"https://pith.science/pith/PLG27OOU3UVG4IRSLFODCMOSNW/action/replication_record"}},"created_at":"2026-07-05T10:51:23.835294+00:00","updated_at":"2026-07-05T10:51:23.835294+00:00"}