{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:A6NLP5DMIOIRHFJ6VI4N74PA6P","short_pith_number":"pith:A6NLP5DM","schema_version":"1.0","canonical_sha256":"079ab7f46c439113953eaa38dff1e0f3f95269cba490a98483bd0ec7c05000bb","source":{"kind":"arxiv","id":"2406.10981","version":1},"attestation_state":"computed","paper":{"title":"ViD-GPT: Introducing GPT-style Autoregressive Generation in Video Diffusion Models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chunping Wang, Hanwang Zhang, Jiaxin Shi, Jun Xiao, Kaifeng Gao","submitted_at":"2024-06-16T15:37:22Z","abstract_excerpt":"With the advance of diffusion models, today's video generation has achieved impressive quality. But generating temporal consistent long videos is still challenging. A majority of video diffusion models (VDMs) generate long videos in an autoregressive manner, i.e., generating subsequent clips conditioned on last frames of previous clip. However, existing approaches all involve bidirectional computations, which restricts the receptive context of each autoregression step, and results in the model lacking long-term dependencies. Inspired from the huge success of large language models (LLMs) and fo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.10981","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-06-16T15:37:22Z","cross_cats_sorted":[],"title_canon_sha256":"d22a5a13567771b7d641a07a27277026efc60e94bcd980c0377413bd92d4430c","abstract_canon_sha256":"18d9a1d7a42af3587b68c6008ac260b1ca5fe24a28a54d9e176325795f9c5875"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:32:47.238847Z","signature_b64":"uT8mS4rW00u+pVQISa68T08K4A67DZHdLFoAYkd3PgYpqeAIdo0URyQLF8Semk/N2ILo3LG8Jeoyhe/BdnrcCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"079ab7f46c439113953eaa38dff1e0f3f95269cba490a98483bd0ec7c05000bb","last_reissued_at":"2026-07-05T08:32:47.238352Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:32:47.238352Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ViD-GPT: Introducing GPT-style Autoregressive Generation in Video Diffusion Models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chunping Wang, Hanwang Zhang, Jiaxin Shi, Jun Xiao, Kaifeng Gao","submitted_at":"2024-06-16T15:37:22Z","abstract_excerpt":"With the advance of diffusion models, today's video generation has achieved impressive quality. But generating temporal consistent long videos is still challenging. A majority of video diffusion models (VDMs) generate long videos in an autoregressive manner, i.e., generating subsequent clips conditioned on last frames of previous clip. However, existing approaches all involve bidirectional computations, which restricts the receptive context of each autoregression step, and results in the model lacking long-term dependencies. Inspired from the huge success of large language models (LLMs) and fo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.10981","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.10981/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.10981","created_at":"2026-07-05T08:32:47.238413+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.10981v1","created_at":"2026-07-05T08:32:47.238413+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.10981","created_at":"2026-07-05T08:32:47.238413+00:00"},{"alias_kind":"pith_short_12","alias_value":"A6NLP5DMIOIR","created_at":"2026-07-05T08:32:47.238413+00:00"},{"alias_kind":"pith_short_16","alias_value":"A6NLP5DMIOIRHFJ6","created_at":"2026-07-05T08:32:47.238413+00:00"},{"alias_kind":"pith_short_8","alias_value":"A6NLP5DM","created_at":"2026-07-05T08:32:47.238413+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07967","citing_title":"DisCo: World Models with Discrete Camera Motion Control","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16054","citing_title":"Ada-Diffuser: Latent-Aware Adaptive Diffusion for Decision-Making","ref_index":293,"is_internal_anchor":false},{"citing_arxiv_id":"2603.00110","citing_title":"Learning Physics from Pretrained Video Models: A Multimodal Continuous and Sequential World Interaction Models for Robotic Manipulation","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2503.00200","citing_title":"Unified Video Action Model","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15911","citing_title":"Efficient Video Diffusion Models: Advancements and Challenges","ref_index":273,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/A6NLP5DMIOIRHFJ6VI4N74PA6P","json":"https://pith.science/pith/A6NLP5DMIOIRHFJ6VI4N74PA6P.json","graph_json":"https://pith.science/api/pith-number/A6NLP5DMIOIRHFJ6VI4N74PA6P/graph.json","events_json":"https://pith.science/api/pith-number/A6NLP5DMIOIRHFJ6VI4N74PA6P/events.json","paper":"https://pith.science/paper/A6NLP5DM"},"agent_actions":{"view_html":"https://pith.science/pith/A6NLP5DMIOIRHFJ6VI4N74PA6P","download_json":"https://pith.science/pith/A6NLP5DMIOIRHFJ6VI4N74PA6P.json","view_paper":"https://pith.science/paper/A6NLP5DM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.10981&json=true","fetch_graph":"https://pith.science/api/pith-number/A6NLP5DMIOIRHFJ6VI4N74PA6P/graph.json","fetch_events":"https://pith.science/api/pith-number/A6NLP5DMIOIRHFJ6VI4N74PA6P/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/A6NLP5DMIOIRHFJ6VI4N74PA6P/action/timestamp_anchor","attest_storage":"https://pith.science/pith/A6NLP5DMIOIRHFJ6VI4N74PA6P/action/storage_attestation","attest_author":"https://pith.science/pith/A6NLP5DMIOIRHFJ6VI4N74PA6P/action/author_attestation","sign_citation":"https://pith.science/pith/A6NLP5DMIOIRHFJ6VI4N74PA6P/action/citation_signature","submit_replication":"https://pith.science/pith/A6NLP5DMIOIRHFJ6VI4N74PA6P/action/replication_record"}},"created_at":"2026-07-05T08:32:47.238413+00:00","updated_at":"2026-07-05T08:32:47.238413+00:00"}