{"work":{"id":"1b1baf78-58ec-44d0-b700-84dff57b2f1f","openalex_id":null,"doi":null,"arxiv_id":"2308.06571","raw_key":null,"title":"ModelScope Text-to-Video Technical Report","authors":null,"authors_text":"Jiuniu Wang, Hangjie Yuan, Dayou Chen, Yingya Zhang, Xiang Wang, Shiwei Zhang","year":2023,"venue":"cs.CV","abstract":"This paper introduces ModelScopeT2V, a text-to-video synthesis model that evolves from a text-to-image synthesis model (i.e., Stable Diffusion). ModelScopeT2V incorporates spatio-temporal blocks to ensure consistent frame generation and smooth movement transitions. The model could adapt to varying frame numbers during training and inference, rendering it suitable for both image-text and video-text datasets. ModelScopeT2V brings together three components (i.e., VQGAN, a text encoder, and a denoising UNet), totally comprising 1.7 billion parameters, in which 0.5 billion parameters are dedicated to temporal capabilities. The model demonstrates superior performance over state-of-the-art methods across three evaluation metrics. The code and an online demo are available at \\url{https://modelscope.cn/models/damo/text-to-video-synthesis/summary}.","external_url":"https://arxiv.org/abs/2308.06571","cited_by_count":null,"metadata_source":"pith","metadata_fetched_at":"2026-07-11T01:57:50.025885+00:00","pith_arxiv_id":"2308.06571","created_at":"2026-05-09T05:55:29.373552+00:00","updated_at":"2026-07-11T01:57:50.025885+00:00","title_quality_ok":true,"display_title":"ModelScope Text-to-Video Technical Report","render_title":"ModelScope Text-to-Video Technical Report"},"hub":{"state":{"work_id":"1b1baf78-58ec-44d0-b700-84dff57b2f1f","tier":"hub","tier_reason":"10+ Pith inbound or 1,000+ external citations","pith_inbound_count":68,"external_cited_by_count":null,"distinct_field_count":5,"first_pith_cited_at":"2023-10-30T13:12:40+00:00","last_pith_cited_at":"2026-07-07T06:05:42+00:00","author_build_status":"not_needed","summary_status":"needed","contexts_status":"needed","graph_status":"needed","ask_index_status":"not_needed","reader_status":"not_needed","recognition_status":"not_needed","updated_at":"2026-08-23T11:39:25.430621+00:00","tier_text":"hub"},"tier":"hub","role_counts":[{"context_role":"background","n":18},{"context_role":"baseline","n":3},{"context_role":"other","n":1}],"polarity_counts":[{"context_polarity":"background","n":19},{"context_polarity":"baseline","n":3}],"runs":{},"summary":{},"graph":{},"authors":[]}}