{"work":{"id":"5cc3572d-7e2f-4431-ae42-d9282a42a800","openalex_id":"https://openalex.org/W4390137271","doi":"10.48550/arxiv.2312.14125","arxiv_id":"2312.14125","raw_key":null,"title":"VideoPoet: A Large Language Model for Zero-Shot Video Generation","authors":null,"authors_text":"Dan Kondratyuk, Lijun Yu, Xiuye Gu, Jos\\'e Lezama, Jonathan Huang, Grant Schindler","year":2023,"venue":"cs.CV","abstract":"We present VideoPoet, a language model capable of synthesizing high-quality video, with matching audio, from a large variety of conditioning signals. VideoPoet employs a decoder-only transformer architecture that processes multimodal inputs -- including images, videos, text, and audio. The training protocol follows that of Large Language Models (LLMs), consisting of two stages: pretraining and task-specific adaptation. During pretraining, VideoPoet incorporates a mixture of multimodal generative objectives within an autoregressive Transformer framework. The pretrained LLM serves as a foundation that can be adapted for a range of video generation tasks. We present empirical results demonstrating the model's state-of-the-art capabilities in zero-shot video generation, specifically highlighting VideoPoet's ability to generate high-fidelity motions. Project page: http://sites.research.google/videopoet/","external_url":"https://arxiv.org/abs/2312.14125","cited_by_count":19,"metadata_source":"pith","metadata_fetched_at":"2026-08-05T02:28:24.338817+00:00","pith_arxiv_id":"2312.14125","created_at":"2026-05-09T06:40:39.188830+00:00","updated_at":"2026-08-05T02:28:24.338817+00:00","title_quality_ok":true,"display_title":"VideoPoet: A Large Language Model for Zero-Shot Video Generation","render_title":"VideoPoet: A Large Language Model for Zero-Shot Video Generation"},"hub":{"state":{"work_id":"5cc3572d-7e2f-4431-ae42-d9282a42a800","tier":"hub","tier_reason":"10+ Pith inbound or 1,000+ external citations","pith_inbound_count":61,"external_cited_by_count":19,"distinct_field_count":7,"first_pith_cited_at":"2024-02-27T03:30:58+00:00","last_pith_cited_at":"2026-06-30T14:31:32+00:00","author_build_status":"not_needed","summary_status":"needed","contexts_status":"needed","graph_status":"needed","ask_index_status":"not_needed","reader_status":"not_needed","recognition_status":"not_needed","updated_at":"2026-08-21T22:39:30.906228+00:00","tier_text":"hub"},"tier":"hub","role_counts":[{"context_role":"background","n":15},{"context_role":"baseline","n":1}],"polarity_counts":[{"context_polarity":"background","n":15},{"context_polarity":"baseline","n":1}],"runs":{},"summary":{},"graph":{},"authors":[]}}