{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:PKZML3CASBPL24A2GKIQ57HFHM","short_pith_number":"pith:PKZML3CA","schema_version":"1.0","canonical_sha256":"7ab2c5ec40905ebd701a32910efce53b3ebedfc56f5b722b4ddebfa266c4507f","source":{"kind":"arxiv","id":"2410.08151","version":2},"attestation_state":"computed","paper":{"title":"Progressive Autoregressive Video Diffusion Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Arie Kaufman, Desai Xie, Difan Liu, Feng Liu, Hao Tan, Yang Zhou, Yicong Hong, Zhan Xu","submitted_at":"2024-10-10T17:36:15Z","abstract_excerpt":"Current frontier video diffusion models have demonstrated remarkable results at generating high-quality videos. However, they can only generate short video clips, normally around 10 seconds or 240 frames, due to computation limitations during training. Existing methods naively achieve autoregressive long video generation by directly placing the ending of the previous clip at the front of the attention window as conditioning, which leads to abrupt scene changes, unnatural motion, and error accumulation. In this work, we introduce a more natural formulation of autoregressive long video generatio"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.08151","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-10-10T17:36:15Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"6cf60ccab2116482fc46be80b844701b3c1e651f60fac0102aa3290f0b072864","abstract_canon_sha256":"7d0a8c856399a2052435cba81a342d3ff35b7fac1ec9d98727447ab5747b5634"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:04:40.394248Z","signature_b64":"9xSgGa1ekYhmoUaiu6MppXPi0ptJ6jjTxWM4Js+/jznmpvxgbwCXTpJWKjbUvlATZ6GisX+GxHCBzfXpJgyDDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7ab2c5ec40905ebd701a32910efce53b3ebedfc56f5b722b4ddebfa266c4507f","last_reissued_at":"2026-07-05T11:04:40.393772Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:04:40.393772Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Progressive Autoregressive Video Diffusion Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Arie Kaufman, Desai Xie, Difan Liu, Feng Liu, Hao Tan, Yang Zhou, Yicong Hong, Zhan Xu","submitted_at":"2024-10-10T17:36:15Z","abstract_excerpt":"Current frontier video diffusion models have demonstrated remarkable results at generating high-quality videos. However, they can only generate short video clips, normally around 10 seconds or 240 frames, due to computation limitations during training. Existing methods naively achieve autoregressive long video generation by directly placing the ending of the previous clip at the front of the attention window as conditioning, which leads to abrupt scene changes, unnatural motion, and error accumulation. In this work, we introduce a more natural formulation of autoregressive long video generatio"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.08151","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.08151/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.08151","created_at":"2026-07-05T11:04:40.393830+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.08151v2","created_at":"2026-07-05T11:04:40.393830+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.08151","created_at":"2026-07-05T11:04:40.393830+00:00"},{"alias_kind":"pith_short_12","alias_value":"PKZML3CASBPL","created_at":"2026-07-05T11:04:40.393830+00:00"},{"alias_kind":"pith_short_16","alias_value":"PKZML3CASBPL24A2","created_at":"2026-07-05T11:04:40.393830+00:00"},{"alias_kind":"pith_short_8","alias_value":"PKZML3CA","created_at":"2026-07-05T11:04:40.393830+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09828","citing_title":"Latent Spatial Memory for Video World Models","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30083","citing_title":"Future Forcing: Future-aware Training-free KV Cache Policy for Autoregressive Video Generation","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16054","citing_title":"Ada-Diffuser: Latent-Aware Adaptive Diffusion for Decision-Making","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2505.21996","citing_title":"VRAG: Learning World Models for Interactive Video Generation","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2506.08009","citing_title":"Self Forcing: Bridging the Train-Test Gap in Autoregressive Video Diffusion","ref_index":93,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07209","citing_title":"INSPATIO-WORLD: A Real-Time 4D World Simulator via Spatiotemporal Autoregressive Modeling","ref_index":92,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PKZML3CASBPL24A2GKIQ57HFHM","json":"https://pith.science/pith/PKZML3CASBPL24A2GKIQ57HFHM.json","graph_json":"https://pith.science/api/pith-number/PKZML3CASBPL24A2GKIQ57HFHM/graph.json","events_json":"https://pith.science/api/pith-number/PKZML3CASBPL24A2GKIQ57HFHM/events.json","paper":"https://pith.science/paper/PKZML3CA"},"agent_actions":{"view_html":"https://pith.science/pith/PKZML3CASBPL24A2GKIQ57HFHM","download_json":"https://pith.science/pith/PKZML3CASBPL24A2GKIQ57HFHM.json","view_paper":"https://pith.science/paper/PKZML3CA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.08151&json=true","fetch_graph":"https://pith.science/api/pith-number/PKZML3CASBPL24A2GKIQ57HFHM/graph.json","fetch_events":"https://pith.science/api/pith-number/PKZML3CASBPL24A2GKIQ57HFHM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PKZML3CASBPL24A2GKIQ57HFHM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PKZML3CASBPL24A2GKIQ57HFHM/action/storage_attestation","attest_author":"https://pith.science/pith/PKZML3CASBPL24A2GKIQ57HFHM/action/author_attestation","sign_citation":"https://pith.science/pith/PKZML3CASBPL24A2GKIQ57HFHM/action/citation_signature","submit_replication":"https://pith.science/pith/PKZML3CASBPL24A2GKIQ57HFHM/action/replication_record"}},"created_at":"2026-07-05T11:04:40.393830+00:00","updated_at":"2026-07-05T11:04:40.393830+00:00"}