{"total":20,"items":[{"citing_arxiv_id":"2606.29473","ref_index":54,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"MAVIN: Multi-Shot Audio-Visual Generation with Customized Narrative Control","primary_cat":"cs.CV","submitted_at":"2026-06-28T16:01:04+00:00","verdict":"CONDITIONAL","verdict_confidence":"HIGH","novelty_score":7.0,"formal_verification":"none","one_line_summary":"MAVIN is a dual-tower diffusion framework that produces temporally aligned multi-shot audio-visual content from hierarchical captions with optional multi-identity image and audio references.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.03672","ref_index":25,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Foley-Omni: A Unified Multimodal Generation Model from Task-Level Audio Synthesis to Complete Video Soundtrack Generation","primary_cat":"cs.SD","submitted_at":"2026-06-02T13:56:31+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Foley-Omni extends isolated audio synthesis to joint generation of full video soundtracks across speech, effects, and music, with a new V2ST-Bench for evaluation showing competitive single-task results and gains in mixed-track consistency.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.03168","ref_index":22,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"JAVEDIT: Joint Audio-Visual Instruction-Guided Video Editing with Agentic Data Curation","primary_cat":"cs.CV","submitted_at":"2026-06-02T05:26:06+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"JAVEdit-100k is the first large-scale dataset for instruction-guided joint audio-visual video editing, accompanied by JAVEditBench and the JAVEdit model that outperforms baselines on five of six metrics.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.30965","ref_index":59,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"ImmersiveTTS: Environment-Aware Text-to-Speech with Multimodal Diffusion Transformer and Domain-Specific Representation Alignment","primary_cat":"eess.AS","submitted_at":"2026-05-29T07:58:54+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"ImmersiveTTS proposes an environment-aware TTS system that integrates speech with environmental audio via multimodal diffusion transformer, joint attention, and domain-specific representation alignment, claiming superior naturalness and fidelity.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.30339","ref_index":58,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Benchmarking Single-Factor Physical Video-to-Audio Generation","primary_cat":"cs.CV","submitted_at":"2026-05-28T17:59:09+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"FlatSounds benchmark shows state-of-the-art V2A models rely more on text captions than visual input for physical and semantic accuracy, with captions improving correctness but degrading temporal alignment.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.25193","ref_index":11,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"SpongeBob: Sync-Aware Harmonious Audio-Visual Generative Editing","primary_cat":"cs.CV","submitted_at":"2026-05-24T17:50:45+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"SpongeBob introduces the first end-to-end audio-visual joint editing framework using sync-aware bidirectional attention and context-aware modules, plus a new dataset and benchmark, claiming 30% Sync-C and 12.5% Ctx-F1 gains over baselines.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.24652","ref_index":27,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"AVBench: Human-Aligned and Automated Evaluation Benchmark for Audio-Video Generative Models","primary_cat":"cs.AI","submitted_at":"2026-05-23T16:42:39+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"AVBench is a benchmark for human-centric AV generation evaluation featuring ten fine-grained dimensions and preference-learned evaluators that output continuous probabilistic scores from binary decisions.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.18749","ref_index":18,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"WavFlow: Audio Generation in Waveform Space","primary_cat":"cs.SD","submitted_at":"2026-05-18T17:59:10+00:00","verdict":"CONDITIONAL","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"WavFlow performs direct waveform audio generation via flow matching on 2D token grids from raw patches plus amplitude lifting, matching latent-based methods on VGGSound and AudioCaps without intermediate compression.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.18916","ref_index":12,"ref_count":2,"confidence":0.9,"is_internal_anchor":false,"paper_title":"CounterFlow: A Two-Phase Inference-Time Sampling for Counterfactual Video Foley Generation","primary_cat":"cs.MM","submitted_at":"2026-05-18T05:42:06+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"CounterFlow is a dual-phase inference-time sampling scheme for pretrained flow-matching VT2A models that enables generation of counterfactual audio synchronized to video but aligned with a contradictory text prompt.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.17488","ref_index":49,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Omni-Customizer: End-to-End MultiModal Customization for Joint Audio-Video Generation","primary_cat":"cs.CV","submitted_at":"2026-05-17T14:56:52+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Omni-Customizer proposes an end-to-end framework using Omni-Context Fusion, Masked TTS Cross-Attention, Semantic-Anchored Multimodal RoPE, and specialized training curricula to achieve precise multimodal identity binding in joint audio-video generation.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.01809","ref_index":14,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"TMD-Bench: A Multi-Level Evaluation Paradigm for Music-Dance Co-Generation","primary_cat":"cs.SD","submitted_at":"2026-05-03T10:25:47+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"TMD-Bench is a multi-level benchmark that measures music-dance co-generation quality including beat-level rhythmic synchronization, supported by a new dataset and Music Captioner, and shows commercial models lag in rhythm while a new baseline performs competitively.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.15086","ref_index":39,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"ControlFoley: Unified and Controllable Video-to-Audio Generation with Cross-Modal Conflict Handling","primary_cat":"cs.MM","submitted_at":"2026-04-16T14:47:24+00:00","verdict":null,"verdict_confidence":null,"novelty_score":null,"formal_verification":null,"one_line_summary":null,"context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.10542","ref_index":42,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"VidAudio-Bench: Benchmarking V2A and VT2A Generation across Four Audio Categories","primary_cat":"cs.SD","submitted_at":"2026-04-12T09:11:55+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"VidAudio-Bench benchmarks V2A and VT2A models across four audio categories, revealing poor speech/singing performance and a tension between visual alignment and text instruction following.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.09057","ref_index":37,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Tora3: Trajectory-Guided Audio-Video Generation with Physical Coherence","primary_cat":"cs.CV","submitted_at":"2026-04-10T07:37:03+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Tora3 uses shared object trajectories as kinematic priors to jointly guide visual motion and acoustic events in audio-video generation, improving realism and synchronization.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.04348","ref_index":39,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"OmniSonic: Towards Universal and Holistic Audio Generation from Video and Text","primary_cat":"cs.SD","submitted_at":"2026-04-06T01:43:00+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"OmniSonic introduces a TriAttn-DiT architecture with MoE gating to jointly generate on-screen, off-screen, and speech audio from video and text, outperforming prior models on a new UniHAGen-Bench.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2601.02731","ref_index":11,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Omni2Sound: Towards Unified Video-Text-to-Audio Generation","primary_cat":"cs.SD","submitted_at":"2026-01-06T05:49:41+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"A single DiT-based diffusion model unifies video-to-audio, text-to-audio, and joint video-text-to-audio generation, supported by a new 470k-pair dataset and three-stage progressive training that resolves task competition.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2512.23994","ref_index":32,"ref_count":2,"confidence":0.9,"is_internal_anchor":false,"paper_title":"PhyAVBench: A Challenging Audio Physics-Sensitivity Benchmark for Physically Grounded Text-to-Audio-Video Generation","primary_cat":"cs.SD","submitted_at":"2025-12-30T05:22:31+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"PhyAVBench provides the first systematic benchmark and metric for audio-physics grounding in T2AV, I2AV, and V2A models using controlled prompt pairs and real video ground truth.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2512.10571","ref_index":54,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"AVI-Edit: Audio-sync Video Instance Editing with Granularity-Aware Mask Refiner","primary_cat":"cs.CV","submitted_at":"2025-12-11T11:58:53+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"AVI-Edit enables precise audio-synchronized instance-level video editing via a granularity-aware mask refiner, a self-feedback audio agent, and a new large-scale annotated dataset.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2512.09299","ref_index":39,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"VABench: A Comprehensive Benchmark for Audio-Video Generation","primary_cat":"cs.CV","submitted_at":"2025-12-10T03:57:29+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"VABench is a new multi-dimensional benchmark for evaluating synchronous audio-video generation across text-to-AV, image-to-AV, and stereo tasks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2512.00336","ref_index":42,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"MVAD: A Benchmark Dataset for Multimodal AI-Generated Video-Audio Detection","primary_cat":"cs.CV","submitted_at":"2025-11-29T05:59:38+00:00","verdict":"CONDITIONAL","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"MVAD is the first comprehensive benchmark dataset for AI-generated multimodal video-audio detection, with three realistic forgery patterns, high-quality outputs from state-of-the-art models, and diversity across visual styles and content categories.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null}],"limit":50,"offset":0}