{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:G5ILXTEHRFZAB7INVLY4NHPO3C","short_pith_number":"pith:G5ILXTEH","schema_version":"1.0","canonical_sha256":"3750bbcc87897200fd0daaf1c69deed89e9b6124a5cd7a28a2a4b6ddd5dedcff","source":{"kind":"arxiv","id":"2204.03458","version":2},"attestation_state":"computed","paper":{"title":"Video Diffusion Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"A diffusion model extended from images generates high-fidelity coherent videos using joint training and conditional sampling.","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Alexey Gritsenko, David J. Fleet, Jonathan Ho, Mohammad Norouzi, Tim Salimans, William Chan","submitted_at":"2022-04-07T14:08:02Z","abstract_excerpt":"Generating temporally coherent high fidelity video is an important milestone in generative modeling research. We make progress towards this milestone by proposing a diffusion model for video generation that shows very promising initial results. Our model is a natural extension of the standard image diffusion architecture, and it enables jointly training from image and video data, which we find to reduce the variance of minibatch gradients and speed up optimization. To generate long and higher resolution videos we introduce a new conditional sampling technique for spatial and temporal video ext"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":true,"formal_links_present":true},"canonical_record":{"source":{"id":"2204.03458","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2022-04-07T14:08:02Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"298b61f8bcb493c054d01d107797de5bc9d7b8fc05e162952a1b302faddb38f4","abstract_canon_sha256":"5087b293a4582c56e830cacecfb1fbff56e5337a2d10670ddffcdcae3ccc649a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:34:16.382372Z","signature_b64":"Kes47nWfx8U2Xw3gzlWhg6gMwbWRDns0e380ye0/cXe8vdwgX2iX20NHaVeb8ushJ6FlZAGv3HuwkCzhGH6ECQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3750bbcc87897200fd0daaf1c69deed89e9b6124a5cd7a28a2a4b6ddd5dedcff","last_reissued_at":"2026-07-05T04:34:16.381834Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:34:16.381834Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Video Diffusion Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"A diffusion model extended from images generates high-fidelity coherent videos using joint training and conditional sampling.","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Alexey Gritsenko, David J. Fleet, Jonathan Ho, Mohammad Norouzi, Tim Salimans, William Chan","submitted_at":"2022-04-07T14:08:02Z","abstract_excerpt":"Generating temporally coherent high fidelity video is an important milestone in generative modeling research. We make progress towards this milestone by proposing a diffusion model for video generation that shows very promising initial results. Our model is a natural extension of the standard image diffusion architecture, and it enables jointly training from image and video data, which we find to reduce the variance of minibatch gradients and speed up optimization. To generate long and higher resolution videos we introduce a new conditional sampling technique for spatial and temporal video ext"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"We present the first results on a large text-conditioned video generation task, as well as state-of-the-art results on established benchmarks for video prediction and unconditional video generation.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"That treating video as an extension of image diffusion (with joint training and the new conditional sampling) is sufficient to produce temporally coherent high-fidelity output without major additional architectural changes for motion modeling.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"A diffusion model for video generation extends image architectures with joint image-video training and improved conditional sampling, delivering first large-scale text-to-video results and state-of-the-art performance on video prediction and unconditional generation benchmarks.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"A diffusion model extended from images generates high-fidelity coherent videos using joint training and conditional sampling.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"8d06f8360d8bdeda9a262587fa3a9bac87f39b0d32cc5828de52f356c463846e"},"source":{"id":"2204.03458","kind":"arxiv","version":2},"verdict":{"id":"ebab4cc6-dc49-4551-bbc1-bfab60115530","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-13T14:33:29.620042Z","strongest_claim":"We present the first results on a large text-conditioned video generation task, as well as state-of-the-art results on established benchmarks for video prediction and unconditional video generation.","one_line_summary":"A diffusion model for video generation extends image architectures with joint image-video training and improved conditional sampling, delivering first large-scale text-to-video results and state-of-the-art performance on video prediction and unconditional generation benchmarks.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"That treating video as an extension of image diffusion (with joint training and the new conditional sampling) is sufficient to produce temporally coherent high-fidelity output without major additional architectural changes for motion modeling.","pith_extraction_headline":"A diffusion model extended from images generates high-fidelity coherent videos using joint training and conditional sampling."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2204.03458/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":65,"sample":[{"doi":"","year":2022,"title":"https://www.tensorflow.org/ datasets","work_id":"fe8e1ac2-0b6d-4ece-ab8b-95700da9973b","ref_index":1,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2021,"title":"ViViT: A video vision transformer","work_id":"020ac0e6-07b7-4ecb-af98-6915815dddc1","ref_index":2,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2017,"title":"Stochastic Variational Video Prediction","work_id":"2b4f01f7-2946-42ed-ad06-677913824304","ref_index":3,"cited_arxiv_id":"1710.11252","is_internal_anchor":false},{"doi":"","year":2021,"title":"Fitvid: Overfitting in pixel-level video prediction.arXiv preprint arXiv:2106.13195","work_id":"98b75ffa-1d61-4641-a59f-5967267b7d2c","ref_index":4,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2021,"title":"Is space-time attention all you need for video understanding?","work_id":"02f6f42d-c731-4407-ba8f-b5c8d7c0d938","ref_index":5,"cited_arxiv_id":"","is_internal_anchor":false}],"resolved_work":65,"snapshot_sha256":"d7eae8114c19f1ccb3d7231ec99a4a73d7cda6c83a748273111d818fd01f2648","internal_anchors":5},"formal_canon":{"evidence_count":2,"snapshot_sha256":"bdcc741e60defc283a1933b32193070ec10fe166c6013642255e2acdcb028868"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2204.03458","created_at":"2026-07-05T04:34:16.381919+00:00"},{"alias_kind":"arxiv_version","alias_value":"2204.03458v2","created_at":"2026-07-05T04:34:16.381919+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2204.03458","created_at":"2026-07-05T04:34:16.381919+00:00"},{"alias_kind":"pith_short_12","alias_value":"G5ILXTEHRFZA","created_at":"2026-07-05T04:34:16.381919+00:00"},{"alias_kind":"pith_short_16","alias_value":"G5ILXTEHRFZAB7IN","created_at":"2026-07-05T04:34:16.381919+00:00"},{"alias_kind":"pith_short_8","alias_value":"G5ILXTEH","created_at":"2026-07-05T04:34:16.381919+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":65,"internal_anchor_count":65,"sample":[{"citing_arxiv_id":"2606.06497","citing_title":"Real-Time AttentionBender: Granular Interactive Network Bending of Video Diffusion Transformers","ref_index":7,"is_internal_anchor":true},{"citing_arxiv_id":"2605.00412","citing_title":"Physically Native World Models: A Hamiltonian Perspective on Generative World Modeling","ref_index":10,"is_internal_anchor":true},{"citing_arxiv_id":"2607.01775","citing_title":"Set Diffusion: Interpolating Token Orderings Between Autoregression and Diffusion for Fast and Flexible Decoding","ref_index":90,"is_internal_anchor":true},{"citing_arxiv_id":"2606.10671","citing_title":"FadeMem: Distance-Aware Memory Consolidation for Autoregressive Video Diffusion","ref_index":48,"is_internal_anchor":true},{"citing_arxiv_id":"2606.30514","citing_title":"3D Scene-Adaptive Trajectory-Controllable Human Image Animation with Camera Movement","ref_index":9,"is_internal_anchor":true},{"citing_arxiv_id":"2606.02919","citing_title":"Pixel Cube: Diffusion-based Portrait Video Relighting Through Realistic Lighting Reproduction","ref_index":12,"is_internal_anchor":true},{"citing_arxiv_id":"2606.02491","citing_title":"MORPHOS: Autoregressive 4D Generation with Temporal Structured Latents","ref_index":15,"is_internal_anchor":true},{"citing_arxiv_id":"2606.02241","citing_title":"BlockGen: Flexible Blockwise Sequence Modeling with Hybrid Samplers","ref_index":127,"is_internal_anchor":true},{"citing_arxiv_id":"2606.31576","citing_title":"Introduction to Stochastic Differential Equations for Generative Machine Learning: A Variational Perspective","ref_index":7,"is_internal_anchor":true},{"citing_arxiv_id":"2605.00412","citing_title":"Physically Native World Models: A Hamiltonian Perspective on Generative World Modeling","ref_index":10,"is_internal_anchor":true},{"citing_arxiv_id":"2605.15116","citing_title":"DriveCtrl: Conditioned Sim-to-Real Driving Video Generation","ref_index":20,"is_internal_anchor":true},{"citing_arxiv_id":"2605.24509","citing_title":"{\\Phi}-Noise: Training-Free Temporal Video Conditioning via Phase-Based Noise Manipulation","ref_index":23,"is_internal_anchor":true},{"citing_arxiv_id":"2606.30514","citing_title":"3D Scene-Adaptive Trajectory-Controllable Human Image Animation with Camera Movement","ref_index":9,"is_internal_anchor":true},{"citing_arxiv_id":"2605.25550","citing_title":"DisagFusion: Asynchronous Pipeline Parallelism and Elastic Scheduling for Disaggregated Diffusion Serving","ref_index":8,"is_internal_anchor":true},{"citing_arxiv_id":"2606.07568","citing_title":"A Systematic Study of Behavioral Cloning for Scientific Data Annotation","ref_index":266,"is_internal_anchor":true},{"citing_arxiv_id":"2606.00583","citing_title":"Improving Visual Representation Alignment Generation with GRPO","ref_index":10,"is_internal_anchor":true},{"citing_arxiv_id":"2606.00658","citing_title":"Collaborative Few-Step Distillation and Low-Bit Quantization for Wan2.2 Dual-Expert Video Diffusion Models","ref_index":3,"is_internal_anchor":true},{"citing_arxiv_id":"2606.08674","citing_title":"BioVid: Autoregressive Video Generation with Biological Behavior Semantic Comprehension","ref_index":2,"is_internal_anchor":true},{"citing_arxiv_id":"2311.04938","citing_title":"Improved DDIM Sampling with Moment Matching Gaussian Mixtures","ref_index":6,"is_internal_anchor":true},{"citing_arxiv_id":"2412.15689","citing_title":"DOLLAR: Few-Step Video Generation via Distillation and Latent Reward Optimization","ref_index":17,"is_internal_anchor":true},{"citing_arxiv_id":"2503.20314","citing_title":"Wan: Open and Advanced Large-Scale Video Generative Models","ref_index":17,"is_internal_anchor":true},{"citing_arxiv_id":"2504.17180","citing_title":"We'll Fix it in Post: Improving Text-to-Video Generation with Neuro-Symbolic Feedback","ref_index":32,"is_internal_anchor":true},{"citing_arxiv_id":"2504.18576","citing_title":"DriVerse: Navigation World Model for Driving Simulation via Multimodal Trajectory Prompting and Motion Alignment","ref_index":27,"is_internal_anchor":true},{"citing_arxiv_id":"2605.18010","citing_title":"Functionalization via Structure Completion and Motion Rectification","ref_index":118,"is_internal_anchor":true},{"citing_arxiv_id":"2405.10314","citing_title":"CAT3D: Create Anything in 3D with Multi-View Diffusion Models","ref_index":48,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":2,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/G5ILXTEHRFZAB7INVLY4NHPO3C","json":"https://pith.science/pith/G5ILXTEHRFZAB7INVLY4NHPO3C.json","graph_json":"https://pith.science/api/pith-number/G5ILXTEHRFZAB7INVLY4NHPO3C/graph.json","events_json":"https://pith.science/api/pith-number/G5ILXTEHRFZAB7INVLY4NHPO3C/events.json","paper":"https://pith.science/paper/G5ILXTEH"},"agent_actions":{"view_html":"https://pith.science/pith/G5ILXTEHRFZAB7INVLY4NHPO3C","download_json":"https://pith.science/pith/G5ILXTEHRFZAB7INVLY4NHPO3C.json","view_paper":"https://pith.science/paper/G5ILXTEH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2204.03458&json=true","fetch_graph":"https://pith.science/api/pith-number/G5ILXTEHRFZAB7INVLY4NHPO3C/graph.json","fetch_events":"https://pith.science/api/pith-number/G5ILXTEHRFZAB7INVLY4NHPO3C/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/G5ILXTEHRFZAB7INVLY4NHPO3C/action/timestamp_anchor","attest_storage":"https://pith.science/pith/G5ILXTEHRFZAB7INVLY4NHPO3C/action/storage_attestation","attest_author":"https://pith.science/pith/G5ILXTEHRFZAB7INVLY4NHPO3C/action/author_attestation","sign_citation":"https://pith.science/pith/G5ILXTEHRFZAB7INVLY4NHPO3C/action/citation_signature","submit_replication":"https://pith.science/pith/G5ILXTEHRFZAB7INVLY4NHPO3C/action/replication_record"}},"created_at":"2026-07-05T04:34:16.381919+00:00","updated_at":"2026-07-05T04:34:16.381919+00:00"}