{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2022:G5ILXTEHRFZAB7INVLY4NHPO3C","short_pith_number":"pith:G5ILXTEH","canonical_record":{"source":{"id":"2204.03458","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2022-04-07T14:08:02Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"298b61f8bcb493c054d01d107797de5bc9d7b8fc05e162952a1b302faddb38f4","abstract_canon_sha256":"5087b293a4582c56e830cacecfb1fbff56e5337a2d10670ddffcdcae3ccc649a"},"schema_version":"1.0"},"canonical_sha256":"3750bbcc87897200fd0daaf1c69deed89e9b6124a5cd7a28a2a4b6ddd5dedcff","source":{"kind":"arxiv","id":"2204.03458","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2204.03458","created_at":"2026-07-05T04:34:16Z"},{"alias_kind":"arxiv_version","alias_value":"2204.03458v2","created_at":"2026-07-05T04:34:16Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2204.03458","created_at":"2026-07-05T04:34:16Z"},{"alias_kind":"pith_short_12","alias_value":"G5ILXTEHRFZA","created_at":"2026-07-05T04:34:16Z"},{"alias_kind":"pith_short_16","alias_value":"G5ILXTEHRFZAB7IN","created_at":"2026-07-05T04:34:16Z"},{"alias_kind":"pith_short_8","alias_value":"G5ILXTEH","created_at":"2026-07-05T04:34:16Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2022:G5ILXTEHRFZAB7INVLY4NHPO3C","target":"record","payload":{"canonical_record":{"source":{"id":"2204.03458","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2022-04-07T14:08:02Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"298b61f8bcb493c054d01d107797de5bc9d7b8fc05e162952a1b302faddb38f4","abstract_canon_sha256":"5087b293a4582c56e830cacecfb1fbff56e5337a2d10670ddffcdcae3ccc649a"},"schema_version":"1.0"},"canonical_sha256":"3750bbcc87897200fd0daaf1c69deed89e9b6124a5cd7a28a2a4b6ddd5dedcff","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:34:16.382372Z","signature_b64":"Kes47nWfx8U2Xw3gzlWhg6gMwbWRDns0e380ye0/cXe8vdwgX2iX20NHaVeb8ushJ6FlZAGv3HuwkCzhGH6ECQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3750bbcc87897200fd0daaf1c69deed89e9b6124a5cd7a28a2a4b6ddd5dedcff","last_reissued_at":"2026-07-05T04:34:16.381834Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:34:16.381834Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2204.03458","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T04:34:16Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"HA8IymTQbAIf0kQHtb77a7S1X8XXxVOhcGpsdzuR/qEKwE3yd70na/DQCdQv1yjJauKD2uM9i7Q/KcSVD4/gCA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-17T22:46:31.205829Z"},"content_sha256":"758c6d2b09e26d2fe9a4275ebdead24d861716329a2837532d28fdfa44851235","schema_version":"1.0","event_id":"sha256:758c6d2b09e26d2fe9a4275ebdead24d861716329a2837532d28fdfa44851235"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2022:G5ILXTEHRFZAB7INVLY4NHPO3C","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Video Diffusion Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"A diffusion model extended from images generates high-fidelity coherent videos using joint training and conditional sampling.","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Alexey Gritsenko, David J. Fleet, Jonathan Ho, Mohammad Norouzi, Tim Salimans, William Chan","submitted_at":"2022-04-07T14:08:02Z","abstract_excerpt":"Generating temporally coherent high fidelity video is an important milestone in generative modeling research. We make progress towards this milestone by proposing a diffusion model for video generation that shows very promising initial results. Our model is a natural extension of the standard image diffusion architecture, and it enables jointly training from image and video data, which we find to reduce the variance of minibatch gradients and speed up optimization. To generate long and higher resolution videos we introduce a new conditional sampling technique for spatial and temporal video ext"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"We present the first results on a large text-conditioned video generation task, as well as state-of-the-art results on established benchmarks for video prediction and unconditional video generation.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"That treating video as an extension of image diffusion (with joint training and the new conditional sampling) is sufficient to produce temporally coherent high-fidelity output without major additional architectural changes for motion modeling.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"A diffusion model for video generation extends image architectures with joint image-video training and improved conditional sampling, delivering first large-scale text-to-video results and state-of-the-art performance on video prediction and unconditional generation benchmarks.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"A diffusion model extended from images generates high-fidelity coherent videos using joint training and conditional sampling.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"8d06f8360d8bdeda9a262587fa3a9bac87f39b0d32cc5828de52f356c463846e"},"source":{"id":"2204.03458","kind":"arxiv","version":2},"verdict":{"id":"ebab4cc6-dc49-4551-bbc1-bfab60115530","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-13T14:33:29.620042Z","strongest_claim":"We present the first results on a large text-conditioned video generation task, as well as state-of-the-art results on established benchmarks for video prediction and unconditional video generation.","one_line_summary":"A diffusion model for video generation extends image architectures with joint image-video training and improved conditional sampling, delivering first large-scale text-to-video results and state-of-the-art performance on video prediction and unconditional generation benchmarks.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"That treating video as an extension of image diffusion (with joint training and the new conditional sampling) is sufficient to produce temporally coherent high-fidelity output without major additional architectural changes for motion modeling.","pith_extraction_headline":"A diffusion model extended from images generates high-fidelity coherent videos using joint training and conditional sampling."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2204.03458/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":65,"sample":[{"doi":"","year":2022,"title":"https://www.tensorflow.org/ datasets","work_id":"fe8e1ac2-0b6d-4ece-ab8b-95700da9973b","ref_index":1,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2021,"title":"ViViT: A video vision transformer","work_id":"020ac0e6-07b7-4ecb-af98-6915815dddc1","ref_index":2,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2017,"title":"Stochastic Variational Video Prediction","work_id":"2b4f01f7-2946-42ed-ad06-677913824304","ref_index":3,"cited_arxiv_id":"1710.11252","is_internal_anchor":false},{"doi":"","year":2021,"title":"Fitvid: Overfitting in pixel-level video prediction.arXiv preprint arXiv:2106.13195","work_id":"98b75ffa-1d61-4641-a59f-5967267b7d2c","ref_index":4,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2021,"title":"Is space-time attention all you need for video understanding?","work_id":"02f6f42d-c731-4407-ba8f-b5c8d7c0d938","ref_index":5,"cited_arxiv_id":"","is_internal_anchor":false}],"resolved_work":65,"snapshot_sha256":"d7eae8114c19f1ccb3d7231ec99a4a73d7cda6c83a748273111d818fd01f2648","internal_anchors":5},"formal_canon":{"evidence_count":2,"snapshot_sha256":"bdcc741e60defc283a1933b32193070ec10fe166c6013642255e2acdcb028868"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"ebab4cc6-dc49-4551-bbc1-bfab60115530"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T04:34:16Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"E/OuYTo25FbHRVW4SLH6NmLc3yN4xUFjpzr/nQBLjvwuZxCmtpbko7gWLtpmyLipQPoFJiNY39wii3dbj0ruAg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-17T22:46:31.207077Z"},"content_sha256":"cc909dc1d72540452b645dc140546fb40ddb53093924850309b987289ebeb5b5","schema_version":"1.0","event_id":"sha256:cc909dc1d72540452b645dc140546fb40ddb53093924850309b987289ebeb5b5"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/G5ILXTEHRFZAB7INVLY4NHPO3C/bundle.json","state_url":"https://pith.science/pith/G5ILXTEHRFZAB7INVLY4NHPO3C/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/G5ILXTEHRFZAB7INVLY4NHPO3C/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-17T22:46:31Z","links":{"resolver":"https://pith.science/pith/G5ILXTEHRFZAB7INVLY4NHPO3C","bundle":"https://pith.science/pith/G5ILXTEHRFZAB7INVLY4NHPO3C/bundle.json","state":"https://pith.science/pith/G5ILXTEHRFZAB7INVLY4NHPO3C/state.json","well_known_bundle":"https://pith.science/.well-known/pith/G5ILXTEHRFZAB7INVLY4NHPO3C/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2022:G5ILXTEHRFZAB7INVLY4NHPO3C","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"5087b293a4582c56e830cacecfb1fbff56e5337a2d10670ddffcdcae3ccc649a","cross_cats_sorted":["cs.AI","cs.LG"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2022-04-07T14:08:02Z","title_canon_sha256":"298b61f8bcb493c054d01d107797de5bc9d7b8fc05e162952a1b302faddb38f4"},"schema_version":"1.0","source":{"id":"2204.03458","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2204.03458","created_at":"2026-07-05T04:34:16Z"},{"alias_kind":"arxiv_version","alias_value":"2204.03458v2","created_at":"2026-07-05T04:34:16Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2204.03458","created_at":"2026-07-05T04:34:16Z"},{"alias_kind":"pith_short_12","alias_value":"G5ILXTEHRFZA","created_at":"2026-07-05T04:34:16Z"},{"alias_kind":"pith_short_16","alias_value":"G5ILXTEHRFZAB7IN","created_at":"2026-07-05T04:34:16Z"},{"alias_kind":"pith_short_8","alias_value":"G5ILXTEH","created_at":"2026-07-05T04:34:16Z"}],"graph_snapshots":[{"event_id":"sha256:cc909dc1d72540452b645dc140546fb40ddb53093924850309b987289ebeb5b5","target":"graph","created_at":"2026-07-05T04:34:16Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"We present the first results on a large text-conditioned video generation task, as well as state-of-the-art results on established benchmarks for video prediction and unconditional video generation."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"That treating video as an extension of image diffusion (with joint training and the new conditional sampling) is sufficient to produce temporally coherent high-fidelity output without major additional architectural changes for motion modeling."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"A diffusion model for video generation extends image architectures with joint image-video training and improved conditional sampling, delivering first large-scale text-to-video results and state-of-the-art performance on video prediction and unconditional generation benchmarks."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"A diffusion model extended from images generates high-fidelity coherent videos using joint training and conditional sampling."}],"snapshot_sha256":"8d06f8360d8bdeda9a262587fa3a9bac87f39b0d32cc5828de52f356c463846e"},"formal_canon":{"evidence_count":2,"snapshot_sha256":"bdcc741e60defc283a1933b32193070ec10fe166c6013642255e2acdcb028868"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2204.03458/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Generating temporally coherent high fidelity video is an important milestone in generative modeling research. We make progress towards this milestone by proposing a diffusion model for video generation that shows very promising initial results. Our model is a natural extension of the standard image diffusion architecture, and it enables jointly training from image and video data, which we find to reduce the variance of minibatch gradients and speed up optimization. To generate long and higher resolution videos we introduce a new conditional sampling technique for spatial and temporal video ext","authors_text":"Alexey Gritsenko, David J. Fleet, Jonathan Ho, Mohammad Norouzi, Tim Salimans, William Chan","cross_cats":["cs.AI","cs.LG"],"headline":"A diffusion model extended from images generates high-fidelity coherent videos using joint training and conditional sampling.","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2022-04-07T14:08:02Z","title":"Video Diffusion Models"},"references":{"count":65,"internal_anchors":5,"resolved_work":65,"sample":[{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":1,"title":"https://www.tensorflow.org/ datasets","work_id":"fe8e1ac2-0b6d-4ece-ab8b-95700da9973b","year":2022},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":2,"title":"ViViT: A video vision transformer","work_id":"020ac0e6-07b7-4ecb-af98-6915815dddc1","year":2021},{"cited_arxiv_id":"1710.11252","doi":"","is_internal_anchor":false,"ref_index":3,"title":"Stochastic Variational Video Prediction","work_id":"2b4f01f7-2946-42ed-ad06-677913824304","year":2017},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":4,"title":"Fitvid: Overfitting in pixel-level video prediction.arXiv preprint arXiv:2106.13195","work_id":"98b75ffa-1d61-4641-a59f-5967267b7d2c","year":2021},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":5,"title":"Is space-time attention all you need for video understanding?","work_id":"02f6f42d-c731-4407-ba8f-b5c8d7c0d938","year":2021}],"snapshot_sha256":"d7eae8114c19f1ccb3d7231ec99a4a73d7cda6c83a748273111d818fd01f2648"},"source":{"id":"2204.03458","kind":"arxiv","version":2},"verdict":{"created_at":"2026-05-13T14:33:29.620042Z","id":"ebab4cc6-dc49-4551-bbc1-bfab60115530","model_set":{"reader":"grok-4.3"},"one_line_summary":"A diffusion model for video generation extends image architectures with joint image-video training and improved conditional sampling, delivering first large-scale text-to-video results and state-of-the-art performance on video prediction and unconditional generation benchmarks.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"A diffusion model extended from images generates high-fidelity coherent videos using joint training and conditional sampling.","strongest_claim":"We present the first results on a large text-conditioned video generation task, as well as state-of-the-art results on established benchmarks for video prediction and unconditional video generation.","weakest_assumption":"That treating video as an extension of image diffusion (with joint training and the new conditional sampling) is sufficient to produce temporally coherent high-fidelity output without major additional architectural changes for motion modeling."}},"verdict_id":"ebab4cc6-dc49-4551-bbc1-bfab60115530"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:758c6d2b09e26d2fe9a4275ebdead24d861716329a2837532d28fdfa44851235","target":"record","created_at":"2026-07-05T04:34:16Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"5087b293a4582c56e830cacecfb1fbff56e5337a2d10670ddffcdcae3ccc649a","cross_cats_sorted":["cs.AI","cs.LG"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2022-04-07T14:08:02Z","title_canon_sha256":"298b61f8bcb493c054d01d107797de5bc9d7b8fc05e162952a1b302faddb38f4"},"schema_version":"1.0","source":{"id":"2204.03458","kind":"arxiv","version":2}},"canonical_sha256":"3750bbcc87897200fd0daaf1c69deed89e9b6124a5cd7a28a2a4b6ddd5dedcff","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"3750bbcc87897200fd0daaf1c69deed89e9b6124a5cd7a28a2a4b6ddd5dedcff","first_computed_at":"2026-07-05T04:34:16.381834Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T04:34:16.381834Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"Kes47nWfx8U2Xw3gzlWhg6gMwbWRDns0e380ye0/cXe8vdwgX2iX20NHaVeb8ushJ6FlZAGv3HuwkCzhGH6ECQ==","signature_status":"signed_v1","signed_at":"2026-07-05T04:34:16.382372Z","signed_message":"canonical_sha256_bytes"},"source_id":"2204.03458","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:758c6d2b09e26d2fe9a4275ebdead24d861716329a2837532d28fdfa44851235","sha256:cc909dc1d72540452b645dc140546fb40ddb53093924850309b987289ebeb5b5"],"state_sha256":"2c02ef592470cec5a98f2f50213f33dd530a8d11eb01ebf4a414bc926ef8c510"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"VsFMNiZQT3UFKyI+/JWBrRBrH0E2U+rlD6ETQeNhWgtUESkqBbIu9jkkXN7Fl5KhwETKryz11ep89kVA9PivAw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-17T22:46:31.219482Z","bundle_sha256":"c9ef15583d18f544cb0c4583b4d9dd1ab83cc9997d0fd0bc8dc171d90cfe2331"}}