{"work":{"id":"4f68eada-27e3-437a-a2fe-6e4ca524d0d3","openalex_id":"https://openalex.org/W4389115686","doi":"10.48550/arxiv.2311.15127","arxiv_id":"2311.15127","raw_key":null,"title":"Stable Video Diffusion: Scaling Latent Video Diffusion Models to Large Datasets","authors":null,"authors_text":"Andreas Blattmann, Tim Dockhorn, Sumith Kulal, Daniel Mendelevitch, Maciej Kilian, Dominik Lorenz","year":2023,"venue":"cs.CV","abstract":"We present Stable Video Diffusion - a latent video diffusion model for high-resolution, state-of-the-art text-to-video and image-to-video generation. Recently, latent diffusion models trained for 2D image synthesis have been turned into generative video models by inserting temporal layers and finetuning them on small, high-quality video datasets. However, training methods in the literature vary widely, and the field has yet to agree on a unified strategy for curating video data. In this paper, we identify and evaluate three different stages for successful training of video LDMs: text-to-image pretraining, video pretraining, and high-quality video finetuning. Furthermore, we demonstrate the necessity of a well-curated pretraining dataset for generating high-quality videos and present a systematic curation process to train a strong base model, including captioning and filtering strategies. We then explore the impact of finetuning our base model on high-quality data and train a text-to-video model that is competitive with closed-source video generation. We also show that our base model provides a powerful motion representation for downstream tasks such as image-to-video generation and adaptability to camera motion-specific LoRA modules. Finally, we demonstrate that our model provides a strong multi-view 3D-prior and can serve as a base to finetune a multi-view diffusion model that jointly generates multiple views of objects in a feedforward fashion, outperforming image-based methods at a fraction of their compute budget. We release code and model weights at https://github.com/Stability-AI/generative-models .","external_url":"https://arxiv.org/abs/2311.15127","cited_by_count":67,"metadata_source":"pith","metadata_fetched_at":"2026-08-05T02:28:24.338817+00:00","pith_arxiv_id":"2311.15127","created_at":"2026-05-09T05:55:29.366200+00:00","updated_at":"2026-08-05T02:28:24.338817+00:00","title_quality_ok":true,"display_title":"Stable Video Diffusion: Scaling Latent Video Diffusion Models to Large Datasets","render_title":"Stable Video Diffusion: Scaling Latent Video Diffusion Models to Large Datasets"},"hub":{"state":{"work_id":"4f68eada-27e3-437a-a2fe-6e4ca524d0d3","tier":"super_hub","tier_reason":"100+ Pith inbound or 10,000+ external citations","pith_inbound_count":414,"external_cited_by_count":67,"distinct_field_count":14,"first_pith_cited_at":"2023-12-21T18:46:41+00:00","last_pith_cited_at":"2026-07-09T17:59:52+00:00","author_build_status":"needed","summary_status":"needed","contexts_status":"needed","graph_status":"needed","ask_index_status":"needed","reader_status":"not_needed","recognition_status":"not_needed","updated_at":"2026-08-21T07:19:27.503039+00:00","tier_text":"super_hub"},"tier":"super_hub","role_counts":[{"context_role":"background","n":72},{"context_role":"baseline","n":9},{"context_role":"method","n":5},{"context_role":"dataset","n":2},{"context_role":"other","n":1}],"polarity_counts":[{"context_polarity":"background","n":72},{"context_polarity":"baseline","n":9},{"context_polarity":"use_method","n":5},{"context_polarity":"unclear","n":2},{"context_polarity":"use_dataset","n":1}],"runs":{"ask_index":{"job_type":"ask_index","status":"succeeded","result":{"title":"Stable Video Diffusion: Scaling Latent Video Diffusion Models to Large Datasets","claims":[{"claim_text":"We present Stable Video Diffusion - a latent video diffusion model for high-resolution, state-of-the-art text-to-video and image-to-video generation. Recently, latent diffusion models trained for 2D image synthesis have been turned into generative video models by inserting temporal layers and finetuning them on small, high-quality video datasets. However, training methods in the literature vary widely, and the field has yet to agree on a unified strategy for curating video data. In this paper, we identify and evaluate three different stages for successful training of video LDMs: text-to-image ","claim_type":"abstract","evidence_strength":"source_metadata"}],"why_cited":"Pith tracks Stable Video Diffusion: Scaling Latent Video Diffusion Models to Large Datasets because it crossed a citation-hub threshold.","role_counts":[]},"error":null,"updated_at":"2026-05-14T00:04:04.283459+00:00"},"author_expand":{"job_type":"author_expand","status":"succeeded","result":{"authors_linked":[{"id":"af305c1f-915b-4035-8a0b-28a18a2bc39f","orcid":null,"display_name":"Andreas Blattmann"},{"id":"e9176499-49ac-4dc5-bd07-ca9c86fc5e76","orcid":null,"display_name":"Tim Dockhorn"},{"id":"5dc10858-1440-4bf6-a515-691e84fcdb34","orcid":null,"display_name":"Sumith Kulal"},{"id":"3116c037-fef9-4702-8769-d97778a05466","orcid":null,"display_name":"Daniel Mendelevitch"},{"id":"8bb3e168-94d1-4e1e-b588-f1dde18446a0","orcid":null,"display_name":"Maciej Kilian"},{"id":"2b89671d-89cd-442f-a731-9155a466b451","orcid":null,"display_name":"Dominik Lorenz"}]},"error":null,"updated_at":"2026-05-14T00:04:00.456673+00:00"},"context_extract":{"job_type":"context_extract","status":"succeeded","result":{"enqueued_papers":25},"error":null,"updated_at":"2026-05-13T23:54:00.764164+00:00"},"graph_features":{"job_type":"graph_features","status":"succeeded","result":{"co_cited":[{"title":"Wan: Open and Advanced Large-Scale Video Generative Models","work_id":"ad3ebc3b-4224-46c9-b61d-bcf135da0a7c","shared_citers":76},{"title":"CogVideoX: Text-to-Video Diffusion Models with An Expert Transformer","work_id":"f38fc088-12aa-4bf4-9ecd-08d3e797ccb7","shared_citers":51},{"title":"HunyuanVideo: A Systematic Framework For Large Video Generative Models","work_id":"881efa7e-7e73-4c66-9cc3-2803e551061c","shared_citers":47},{"title":"AnimateDiff: Animate Your Personalized Text-to-Image Diffusion Models without Specific Tuning","work_id":"1f9d1d3b-a6d6-45a9-9f13-51393c03be8a","shared_citers":28},{"title":"Flow Matching for Generative Modeling","work_id":"6edb71c4-5d64-40af-a394-9757ea051a36","shared_citers":24},{"title":"Towards Accurate Generative Models of Video: A New Metric & Challenges","work_id":"72f42543-17d5-49aa-ba5a-25d67ffbb88a","shared_citers":22},{"title":"Make-A-Video: Text-to-Video Generation without Text-Video Data","work_id":"52a801fc-a707-45a1-a8cd-0d6702f124ab","shared_citers":20},{"title":"Imagen Video: High Definition Video Generation with Diffusion Models","work_id":"bb20d241-dc6f-4b0a-b071-fd43a2cbd57f","shared_citers":19},{"title":"LTX-Video: Realtime Video Latent Diffusion","work_id":"cee5c521-3ce9-466e-a035-1e42f89254f4","shared_citers":18},{"title":"Auto-Encoding Variational Bayes","work_id":"97d95295-30e1-42b4-bbf6-85f0fa4edb44","shared_citers":17},{"title":"Score-Based Generative Modeling through Stochastic Differential Equations","work_id":"d9110e53-a5d4-4794-a4c5-a575e91c31ad","shared_citers":17},{"title":"CogVideo: Large-scale Pretraining for Text-to-Video Generation via Transformers","work_id":"2dbd6bcd-fc98-4fbf-b586-f6d94fe1abd2","shared_citers":16},{"title":"Movie Gen: A Cast of Media Foundation Models","work_id":"a6a118b0-002f-4b19-881f-7f1183e0d7d8","shared_citers":16},{"title":"SDXL: Improving Latent Diffusion Models for High-Resolution Image Synthesis","work_id":"8034c587-fba6-4941-87ba-c98f2ac962cb","shared_citers":16},{"title":"Classifier-Free Diffusion Guidance","work_id":"acf2c588-c088-4a6c-938e-150ad7c666d7","shared_citers":15},{"title":"Cosmos World Foundation Model Platform for Physical AI","work_id":"a2dba24c-318d-476a-8b21-4289c265810c","shared_citers":15},{"title":"CameraCtrl: Enabling Camera Control for Text-to-Video Generation","work_id":"1c05c278-c023-4ef0-a359-25a41f1065eb","shared_citers":14},{"title":"DINOv2: Learning Robust Visual Features without Supervision","work_id":"26b304e5-b54a-4f26-be7e-83299eca52e4","shared_citers":14},{"title":"ModelScope Text-to-Video Technical Report","work_id":"1b1baf78-58ec-44d0-b700-84dff57b2f1f","shared_citers":14},{"title":"Open-Sora: Democratizing Efficient Video Production for All","work_id":"8b29ba7b-3d84-4281-85b7-9eaf905afd7f","shared_citers":14},{"title":"Qwen2.5-VL Technical Report","work_id":"69dffacb-bfe8-442d-be86-48624c60426f","shared_citers":14},{"title":"Videocrafter1: Open diffusion models for high-quality video generation","work_id":"4d4486c5-6317-4d8d-bb5b-3b100d732a83","shared_citers":14},{"title":"Hierarchical Text-Conditional Image Generation with CLIP Latents","work_id":"0c6a768b-70b8-4242-bb0e-459f1008c9fc","shared_citers":13},{"title":"Latte: Latent Diffusion Transformer for Video Generation","work_id":"5328e907-7278-4781-a2bb-c5ef40dc87fb","shared_citers":13}],"time_series":[{"n":13,"year":2024},{"n":8,"year":2025},{"n":112,"year":2026}]},"error":null,"updated_at":"2026-05-13T23:54:04.780566+00:00"},"identity_refresh":{"job_type":"identity_refresh","status":"succeeded","result":{"fixed":1,"items":[{"title":"Qwen3 Technical Report","work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e","resolver":"local_arxiv","confidence":0.98,"old_work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e"}],"errors":[],"attempted":1},"error":null,"updated_at":"2026-05-13T23:54:14.058834+00:00"},"role_polarity":{"job_type":"role_polarity","status":"succeeded","result":{"title":"Stable Video Diffusion: Scaling Latent Video Diffusion Models to Large Datasets","claims":[{"claim_text":"We present Stable Video Diffusion - a latent video diffusion model for high-resolution, state-of-the-art text-to-video and image-to-video generation. Recently, latent diffusion models trained for 2D image synthesis have been turned into generative video models by inserting temporal layers and finetuning them on small, high-quality video datasets. However, training methods in the literature vary widely, and the field has yet to agree on a unified strategy for curating video data. In this paper, we identify and evaluate three different stages for successful training of video LDMs: text-to-image ","claim_type":"abstract","evidence_strength":"source_metadata"}],"why_cited":"Pith tracks Stable Video Diffusion: Scaling Latent Video Diffusion Models to Large Datasets because it crossed a citation-hub threshold.","role_counts":[]},"error":null,"updated_at":"2026-05-13T23:54:04.669674+00:00"},"summary_claims":{"job_type":"summary_claims","status":"succeeded","result":{"title":"Stable Video Diffusion: Scaling Latent Video Diffusion Models to Large Datasets","claims":[{"claim_text":"We present Stable Video Diffusion - a latent video diffusion model for high-resolution, state-of-the-art text-to-video and image-to-video generation. Recently, latent diffusion models trained for 2D image synthesis have been turned into generative video models by inserting temporal layers and finetuning them on small, high-quality video datasets. However, training methods in the literature vary widely, and the field has yet to agree on a unified strategy for curating video data. In this paper, we identify and evaluate three different stages for successful training of video LDMs: text-to-image ","claim_type":"abstract","evidence_strength":"source_metadata"}],"why_cited":"Pith tracks Stable Video Diffusion: Scaling Latent Video Diffusion Models to Large Datasets because it crossed a citation-hub threshold.","role_counts":[]},"error":null,"updated_at":"2026-05-13T23:54:09.361915+00:00"}},"summary":{"title":"Stable Video Diffusion: Scaling Latent Video Diffusion Models to Large Datasets","claims":[{"claim_text":"We present Stable Video Diffusion - a latent video diffusion model for high-resolution, state-of-the-art text-to-video and image-to-video generation. Recently, latent diffusion models trained for 2D image synthesis have been turned into generative video models by inserting temporal layers and finetuning them on small, high-quality video datasets. However, training methods in the literature vary widely, and the field has yet to agree on a unified strategy for curating video data. In this paper, we identify and evaluate three different stages for successful training of video LDMs: text-to-image ","claim_type":"abstract","evidence_strength":"source_metadata"}],"why_cited":"Pith tracks Stable Video Diffusion: Scaling Latent Video Diffusion Models to Large Datasets because it crossed a citation-hub threshold.","role_counts":[]},"graph":{"co_cited":[{"title":"Wan: Open and Advanced Large-Scale Video Generative Models","work_id":"ad3ebc3b-4224-46c9-b61d-bcf135da0a7c","shared_citers":76},{"title":"CogVideoX: Text-to-Video Diffusion Models with An Expert Transformer","work_id":"f38fc088-12aa-4bf4-9ecd-08d3e797ccb7","shared_citers":51},{"title":"HunyuanVideo: A Systematic Framework For Large Video Generative Models","work_id":"881efa7e-7e73-4c66-9cc3-2803e551061c","shared_citers":47},{"title":"AnimateDiff: Animate Your Personalized Text-to-Image Diffusion Models without Specific Tuning","work_id":"1f9d1d3b-a6d6-45a9-9f13-51393c03be8a","shared_citers":28},{"title":"Flow Matching for Generative Modeling","work_id":"6edb71c4-5d64-40af-a394-9757ea051a36","shared_citers":24},{"title":"Towards Accurate Generative Models of Video: A New Metric & Challenges","work_id":"72f42543-17d5-49aa-ba5a-25d67ffbb88a","shared_citers":22},{"title":"Make-A-Video: Text-to-Video Generation without Text-Video Data","work_id":"52a801fc-a707-45a1-a8cd-0d6702f124ab","shared_citers":20},{"title":"Imagen Video: High Definition Video Generation with Diffusion Models","work_id":"bb20d241-dc6f-4b0a-b071-fd43a2cbd57f","shared_citers":19},{"title":"LTX-Video: Realtime Video Latent Diffusion","work_id":"cee5c521-3ce9-466e-a035-1e42f89254f4","shared_citers":18},{"title":"Auto-Encoding Variational Bayes","work_id":"97d95295-30e1-42b4-bbf6-85f0fa4edb44","shared_citers":17},{"title":"Score-Based Generative Modeling through Stochastic Differential Equations","work_id":"d9110e53-a5d4-4794-a4c5-a575e91c31ad","shared_citers":17},{"title":"CogVideo: Large-scale Pretraining for Text-to-Video Generation via Transformers","work_id":"2dbd6bcd-fc98-4fbf-b586-f6d94fe1abd2","shared_citers":16},{"title":"Movie Gen: A Cast of Media Foundation Models","work_id":"a6a118b0-002f-4b19-881f-7f1183e0d7d8","shared_citers":16},{"title":"SDXL: Improving Latent Diffusion Models for High-Resolution Image Synthesis","work_id":"8034c587-fba6-4941-87ba-c98f2ac962cb","shared_citers":16},{"title":"Classifier-Free Diffusion Guidance","work_id":"acf2c588-c088-4a6c-938e-150ad7c666d7","shared_citers":15},{"title":"Cosmos World Foundation Model Platform for Physical AI","work_id":"a2dba24c-318d-476a-8b21-4289c265810c","shared_citers":15},{"title":"CameraCtrl: Enabling Camera Control for Text-to-Video Generation","work_id":"1c05c278-c023-4ef0-a359-25a41f1065eb","shared_citers":14},{"title":"DINOv2: Learning Robust Visual Features without Supervision","work_id":"26b304e5-b54a-4f26-be7e-83299eca52e4","shared_citers":14},{"title":"ModelScope Text-to-Video Technical Report","work_id":"1b1baf78-58ec-44d0-b700-84dff57b2f1f","shared_citers":14},{"title":"Open-Sora: Democratizing Efficient Video Production for All","work_id":"8b29ba7b-3d84-4281-85b7-9eaf905afd7f","shared_citers":14},{"title":"Qwen2.5-VL Technical Report","work_id":"69dffacb-bfe8-442d-be86-48624c60426f","shared_citers":14},{"title":"Videocrafter1: Open diffusion models for high-quality video generation","work_id":"4d4486c5-6317-4d8d-bb5b-3b100d732a83","shared_citers":14},{"title":"Hierarchical Text-Conditional Image Generation with CLIP Latents","work_id":"0c6a768b-70b8-4242-bb0e-459f1008c9fc","shared_citers":13},{"title":"Latte: Latent Diffusion Transformer for Video Generation","work_id":"5328e907-7278-4781-a2bb-c5ef40dc87fb","shared_citers":13}],"time_series":[{"n":13,"year":2024},{"n":8,"year":2025},{"n":112,"year":2026}]},"authors":[{"id":"af305c1f-915b-4035-8a0b-28a18a2bc39f","orcid":null,"display_name":"Andreas Blattmann","source":"manual","import_confidence":0.72},{"id":"3116c037-fef9-4702-8769-d97778a05466","orcid":null,"display_name":"Daniel Mendelevitch","source":"manual","import_confidence":0.72},{"id":"2b89671d-89cd-442f-a731-9155a466b451","orcid":null,"display_name":"Dominik Lorenz","source":"manual","import_confidence":0.72},{"id":"8bb3e168-94d1-4e1e-b588-f1dde18446a0","orcid":null,"display_name":"Maciej Kilian","source":"manual","import_confidence":0.72},{"id":"5dc10858-1440-4bf6-a515-691e84fcdb34","orcid":null,"display_name":"Sumith Kulal","source":"manual","import_confidence":0.72},{"id":"e9176499-49ac-4dc5-bd07-ca9c86fc5e76","orcid":null,"display_name":"Tim Dockhorn","source":"manual","import_confidence":0.72}]}}