{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:3CLHODCFFYXXCVEAEJKVDUPDLK","short_pith_number":"pith:3CLHODCF","schema_version":"1.0","canonical_sha256":"d896770c452e2f715480225551d1e35ab05b5e2841327daa9f21ba50369d9df7","source":{"kind":"arxiv","id":"2501.05763","version":4},"attestation_state":"computed","paper":{"title":"StarGen: A Spatiotemporal Autoregression Framework with Video Diffusion Model for Scalable and Controllable Scene Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Danpeng Chen, Guofeng Zhang, Haomin Liu, Hua Xue, Jialin Liu, Jiaqi Hu, Lei Yang, Nan Wang, Shangjin Zhai, Weijian Xie, Xiaomeng Wang, Zhen Peng, Zhichao Ye","submitted_at":"2025-01-10T07:41:47Z","abstract_excerpt":"Recent advances in large reconstruction and generative models have significantly improved scene reconstruction and novel view generation. However, due to compute limitations, each inference with these large models is confined to a small area, making long-range consistent scene generation challenging. To address this, we propose StarGen, a novel framework that employs a pre-trained video diffusion model in an autoregressive manner for long-range scene generation. The generation of each video clip is conditioned on the 3D warping of spatially adjacent images and the temporally overlapping image "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.05763","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-01-10T07:41:47Z","cross_cats_sorted":[],"title_canon_sha256":"3e16578704b9ca689c04fadfbb30a0408b156b459a899cb2fc6fb04832bd783d","abstract_canon_sha256":"72c4c666524342b4e21e3951d872306dbe61bdaaf163e6ed4c9a1698448eece1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:48:24.129499Z","signature_b64":"SndQXWLZV4ZnEP2remRllUyA5O3gPq49TA9Nrw6amxzXFYCJy8p16zqxt2cMezUET3zA9Hr1n0VDTsNDv1ZrCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d896770c452e2f715480225551d1e35ab05b5e2841327daa9f21ba50369d9df7","last_reissued_at":"2026-07-05T10:48:24.129030Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:48:24.129030Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"StarGen: A Spatiotemporal Autoregression Framework with Video Diffusion Model for Scalable and Controllable Scene Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Danpeng Chen, Guofeng Zhang, Haomin Liu, Hua Xue, Jialin Liu, Jiaqi Hu, Lei Yang, Nan Wang, Shangjin Zhai, Weijian Xie, Xiaomeng Wang, Zhen Peng, Zhichao Ye","submitted_at":"2025-01-10T07:41:47Z","abstract_excerpt":"Recent advances in large reconstruction and generative models have significantly improved scene reconstruction and novel view generation. However, due to compute limitations, each inference with these large models is confined to a small area, making long-range consistent scene generation challenging. To address this, we propose StarGen, a novel framework that employs a pre-trained video diffusion model in an autoregressive manner for long-range scene generation. The generation of each video clip is conditioned on the 3D warping of spatially adjacent images and the temporally overlapping image "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.05763","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.05763/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.05763","created_at":"2026-07-05T10:48:24.129087+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.05763v4","created_at":"2026-07-05T10:48:24.129087+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.05763","created_at":"2026-07-05T10:48:24.129087+00:00"},{"alias_kind":"pith_short_12","alias_value":"3CLHODCFFYXX","created_at":"2026-07-05T10:48:24.129087+00:00"},{"alias_kind":"pith_short_16","alias_value":"3CLHODCFFYXXCVEA","created_at":"2026-07-05T10:48:24.129087+00:00"},{"alias_kind":"pith_short_8","alias_value":"3CLHODCF","created_at":"2026-07-05T10:48:24.129087+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.00345","citing_title":"Pose-Aware Diffusion for 3D Generation","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07209","citing_title":"INSPATIO-WORLD: A Real-Time 4D World Simulator via Spatiotemporal Autoregressive Modeling","ref_index":101,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3CLHODCFFYXXCVEAEJKVDUPDLK","json":"https://pith.science/pith/3CLHODCFFYXXCVEAEJKVDUPDLK.json","graph_json":"https://pith.science/api/pith-number/3CLHODCFFYXXCVEAEJKVDUPDLK/graph.json","events_json":"https://pith.science/api/pith-number/3CLHODCFFYXXCVEAEJKVDUPDLK/events.json","paper":"https://pith.science/paper/3CLHODCF"},"agent_actions":{"view_html":"https://pith.science/pith/3CLHODCFFYXXCVEAEJKVDUPDLK","download_json":"https://pith.science/pith/3CLHODCFFYXXCVEAEJKVDUPDLK.json","view_paper":"https://pith.science/paper/3CLHODCF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.05763&json=true","fetch_graph":"https://pith.science/api/pith-number/3CLHODCFFYXXCVEAEJKVDUPDLK/graph.json","fetch_events":"https://pith.science/api/pith-number/3CLHODCFFYXXCVEAEJKVDUPDLK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3CLHODCFFYXXCVEAEJKVDUPDLK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3CLHODCFFYXXCVEAEJKVDUPDLK/action/storage_attestation","attest_author":"https://pith.science/pith/3CLHODCFFYXXCVEAEJKVDUPDLK/action/author_attestation","sign_citation":"https://pith.science/pith/3CLHODCFFYXXCVEAEJKVDUPDLK/action/citation_signature","submit_replication":"https://pith.science/pith/3CLHODCFFYXXCVEAEJKVDUPDLK/action/replication_record"}},"created_at":"2026-07-05T10:48:24.129087+00:00","updated_at":"2026-07-05T10:48:24.129087+00:00"}