{"work":{"id":"1334671f-ee98-4c53-a1d4-0805281f8d2b","openalex_id":null,"doi":null,"arxiv_id":"2601.03233","raw_key":null,"title":"LTX-2: Efficient Joint Audio-Visual Foundation Model","authors":null,"authors_text":"Yoav HaCohen, Benny Brazowski, Nisan Chiprut, Yaki Bitterman, Andrew Kvochko, Avishai Berkowitz","year":2026,"venue":"cs.CV","abstract":"Recent text-to-video diffusion models can generate compelling video sequences, yet they remain silent -- missing the semantic, emotional, and atmospheric cues that audio provides. We introduce LTX-2, an open-source foundational model capable of generating high-quality, temporally synchronized audiovisual content in a unified manner. LTX-2 consists of an asymmetric dual-stream transformer with a 14B-parameter video stream and a 5B-parameter audio stream, coupled through bidirectional audio-video cross-attention layers with temporal positional embeddings and cross-modality AdaLN for shared timestep conditioning. This architecture enables efficient training and inference of a unified audiovisual model while allocating more capacity for video generation than audio generation. We employ a multilingual text encoder for broader prompt understanding and introduce a modality-aware classifier-free guidance (modality-CFG) mechanism for improved audiovisual alignment and controllability. Beyond generating speech, LTX-2 produces rich, coherent audio tracks that follow the characters, environment, style, and emotion of each scene -- complete with natural background and foley elements. In our evaluations, the model achieves state-of-the-art audiovisual quality and prompt adherence among open-source systems, while delivering results comparable to proprietary models at a fraction of their computational cost and inference time. All model weights and code are publicly released.","external_url":"https://arxiv.org/abs/2601.03233","cited_by_count":null,"metadata_source":"pith","metadata_fetched_at":"2026-07-10T01:46:41.047035+00:00","pith_arxiv_id":"2601.03233","created_at":"2026-05-10T05:20:54.890066+00:00","updated_at":"2026-07-10T01:46:41.047035+00:00","title_quality_ok":true,"display_title":"LTX-2: Efficient Joint Audio-Visual Foundation Model","render_title":"LTX-2: Efficient Joint Audio-Visual Foundation Model"},"hub":{"state":{"work_id":"1334671f-ee98-4c53-a1d4-0805281f8d2b","tier":"hub","tier_reason":"10+ Pith inbound or 1,000+ external citations","pith_inbound_count":66,"external_cited_by_count":null,"distinct_field_count":7,"first_pith_cited_at":"2026-01-29T18:57:13+00:00","last_pith_cited_at":"2026-07-09T17:59:11+00:00","author_build_status":"not_needed","summary_status":"needed","contexts_status":"needed","graph_status":"needed","ask_index_status":"not_needed","reader_status":"not_needed","recognition_status":"not_needed","updated_at":"2026-08-22T17:59:27.347088+00:00","tier_text":"hub"},"tier":"hub","role_counts":[{"context_role":"background","n":18},{"context_role":"baseline","n":2}],"polarity_counts":[{"context_polarity":"background","n":18},{"context_polarity":"baseline","n":2}],"runs":{},"summary":{},"graph":{},"authors":[]}}