{"total":13,"items":[{"citing_arxiv_id":"2606.05121","ref_index":28,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Audio Interaction Model","primary_cat":"cs.SD","submitted_at":"2026-06-03T17:26:11+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Audio-Interaction unifies offline and online audio tasks into one streaming model via the SoundFlow framework and a new 2.6M-item streaming corpus, enabling real-time instruction following and proactive responses.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.03672","ref_index":82,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Foley-Omni: A Unified Multimodal Generation Model from Task-Level Audio Synthesis to Complete Video Soundtrack Generation","primary_cat":"cs.SD","submitted_at":"2026-06-02T13:56:31+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Foley-Omni extends isolated audio synthesis to joint generation of full video soundtracks across speech, effects, and music, with a new V2ST-Bench for evaluation showing competitive single-task results and gains in mixed-track consistency.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.02739","ref_index":29,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"EntangleCodec: A Unified Discrete Audio Tokenizer via Semantic-Acoustic Entanglement","primary_cat":"cs.SD","submitted_at":"2026-06-01T18:05:18+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":4.0,"formal_verification":"none","one_line_summary":"EntangleCodec unifies semantic and acoustic audio tokenization via caption alignment and flow-matching decoding, reporting competitive reconstruction, +7.4% gains on MMAR understanding, and 0.6B-parameter ALMs surpassing 13B-parameter continuous baselines.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.01703","ref_index":20,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"JenBridge: Adaptive Long-Form Video Soundtracking across Scene Transitions","primary_cat":"cs.SD","submitted_at":"2026-06-01T05:12:20+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"JenBridge pretrains a flow-matching Transformer on text-audio data then adapts it with video conditioning and an LLM director to select transitions, claiming better coherence than prior methods on a new LVS benchmark.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.27838","ref_index":17,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Dasheng AudioGen: A Unified Model for Generating Coherent Audio Scenes from Text","primary_cat":"cs.SD","submitted_at":"2026-05-27T01:50:29+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Dasheng AudioGen uses multi-view captions and a unified semantic-acoustic representation to enable end-to-end generation of mixed audio scenes from text descriptions.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.18749","ref_index":19,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"WavFlow: Audio Generation in Waveform Space","primary_cat":"cs.SD","submitted_at":"2026-05-18T17:59:10+00:00","verdict":"CONDITIONAL","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"WavFlow performs direct waveform audio generation via flow matching on 2D token grids from raw patches plus amplitude lifting, matching latent-based methods on VGGSound and AudioCaps without intermediate compression.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.16515","ref_index":45,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"SeamCam: Quantifying Seamless Camouflage via Multi-Cue Visual Detectability","primary_cat":"cs.CV","submitted_at":"2026-05-15T18:08:27+00:00","verdict":"CONDITIONAL","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"SeamCam quantifies camouflage by computing one minus the highest IoU recoverable from category-conditioned detection proposals against a ground-truth mask, achieving 78.82% agreement with human judgments.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.15086","ref_index":40,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"ControlFoley: Unified and Controllable Video-to-Audio Generation with Cross-Modal Conflict Handling","primary_cat":"cs.MM","submitted_at":"2026-04-16T14:47:24+00:00","verdict":null,"verdict_confidence":null,"novelty_score":null,"formal_verification":null,"one_line_summary":null,"context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.14707","ref_index":62,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Geo2Sound: A Scalable Geo-Aligned Framework for Soundscape Generation from Satellite Imagery","primary_cat":"cs.MM","submitted_at":"2026-04-16T07:15:37+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"Geo2Sound generates geographically realistic soundscapes from satellite imagery via geospatial attribute modeling, semantic hypothesis expansion, and geo-acoustic alignment, achieving SOTA FAD of 1.765 on a new 20k-pair benchmark.","context_count":1,"top_context_role":"background","top_context_polarity":"background","context_text":"by Connecting Foundation Models. InProceedings of the AAAI Conference on Artificial Intelligence, Vol. 38. 15492-15501. [61] Junbo Wang, Haofeng Tan, Bowen Liao, Albert Jiang, Teng Fei, Qixing Huang, Bing Zhou, Zhengzhong Tu, Shan Ye, and Yuhao Kang. 2025. SounDiT: Geo-Contextual Soundscape-to-Landscape Generation.arXiv preprint arXiv:2505.12734(2025). [62] Xinyu Wang et al. 2024. TiVA: Time-Aligned Video-to-Audio Generation. In ACM International Conference on Multimedia (ACM MM). [63] Zhenyu Wang, Chenxing Li, Yong Xu, Chunlei Zhang, John H. L. Hansen, and Dong Yu. 2024. Text-to-Audio Generation via Bridging Audio Language Model and Latent Diffusion. InAudio Imagination Workshop at NeurIPS. [64] Zhecheng Wang, Rajanie Prabha, Tianyuan Huang, Jiajun Wu, and Ram Ra-"},{"citing_arxiv_id":"2604.10542","ref_index":48,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"VidAudio-Bench: Benchmarking V2A and VT2A Generation across Four Audio Categories","primary_cat":"cs.SD","submitted_at":"2026-04-12T09:11:55+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"VidAudio-Bench benchmarks V2A and VT2A models across four audio categories, revealing poor speech/singing performance and a tension between visual alignment and text instruction following.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2601.02731","ref_index":14,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Omni2Sound: Towards Unified Video-Text-to-Audio Generation","primary_cat":"cs.SD","submitted_at":"2026-01-06T05:49:41+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"A single DiT-based diffusion model unifies video-to-audio, text-to-audio, and joint video-text-to-audio generation, supported by a new 470k-pair dataset and three-stage progressive training that resolves task competition.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2512.00336","ref_index":44,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"MVAD: A Benchmark Dataset for Multimodal AI-Generated Video-Audio Detection","primary_cat":"cs.CV","submitted_at":"2025-11-29T05:59:38+00:00","verdict":"CONDITIONAL","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"MVAD is the first comprehensive benchmark dataset for AI-generated multimodal video-audio detection, with three realistic forgery patterns, high-quality outputs from state-of-the-art models, and diversity across visual styles and content categories.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2509.18272","ref_index":14,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"StereoFoley: Object-Aware Stereo Audio Generation from Video","primary_cat":"cs.SD","submitted_at":"2025-09-22T18:00:54+00:00","verdict":"CONDITIONAL","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"StereoFoley is an end-to-end video-to-stereo-audio framework that uses a base generative model fine-tuned on synthetic object-tracked data with panning and distance controls to achieve object-aware spatial sound.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null}],"limit":50,"offset":0}