{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:WX2W2TUBI3XNEX7HBTMWTHRRFM","short_pith_number":"pith:WX2W2TUB","schema_version":"1.0","canonical_sha256":"b5f56d4e8146eed25fe70cd9699e312b062a5e447793e69cf584a5da709a10f0","source":{"kind":"arxiv","id":"2503.10719","version":2},"attestation_state":"computed","paper":{"title":"Long-Video Audio Synthesis with Multi-Agent Collaboration","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Li Liu, Xiaojie Xu, Xinli Xu, Yehang Zhang, Yingcong Chen","submitted_at":"2025-03-13T07:58:23Z","abstract_excerpt":"Video-to-audio synthesis, which generates synchronized audio for visual content, critically enhances viewer immersion and narrative coherence in film and interactive media. However, video-to-audio dubbing for long-form content remains an unsolved challenge due to dynamic semantic shifts, temporal misalignment, and the absence of dedicated datasets. While existing methods excel in short videos, they falter in long scenarios (e.g., movies) due to fragmented synthesis and inadequate cross-scene consistency. We propose LVAS-Agent, a novel multi-agent framework that emulates professional dubbing wo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.10719","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CV","submitted_at":"2025-03-13T07:58:23Z","cross_cats_sorted":[],"title_canon_sha256":"b42709840bd32d8062daf6c2dfcb5e179b336e691f4e9b18fdc0aea9587a88ce","abstract_canon_sha256":"6252fa244b34f52c180f65f0805a1bc546546e994a1ca672e509289c1874a695"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:32:41.744284Z","signature_b64":"nqn7GO05R1dZWNQdRvupdtI5DNQSDEmUQcSNcd0E1ER0oWSImR2Kt596uXuZxUllbT4LB+BbMq2sQ/R0tVZ9DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b5f56d4e8146eed25fe70cd9699e312b062a5e447793e69cf584a5da709a10f0","last_reissued_at":"2026-07-05T10:32:41.743340Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:32:41.743340Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Long-Video Audio Synthesis with Multi-Agent Collaboration","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Li Liu, Xiaojie Xu, Xinli Xu, Yehang Zhang, Yingcong Chen","submitted_at":"2025-03-13T07:58:23Z","abstract_excerpt":"Video-to-audio synthesis, which generates synchronized audio for visual content, critically enhances viewer immersion and narrative coherence in film and interactive media. However, video-to-audio dubbing for long-form content remains an unsolved challenge due to dynamic semantic shifts, temporal misalignment, and the absence of dedicated datasets. While existing methods excel in short videos, they falter in long scenarios (e.g., movies) due to fragmented synthesis and inadequate cross-scene consistency. We propose LVAS-Agent, a novel multi-agent framework that emulates professional dubbing wo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.10719","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.10719/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.10719","created_at":"2026-07-05T10:32:41.743463+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.10719v2","created_at":"2026-07-05T10:32:41.743463+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.10719","created_at":"2026-07-05T10:32:41.743463+00:00"},{"alias_kind":"pith_short_12","alias_value":"WX2W2TUBI3XN","created_at":"2026-07-05T10:32:41.743463+00:00"},{"alias_kind":"pith_short_16","alias_value":"WX2W2TUBI3XNEX7H","created_at":"2026-07-05T10:32:41.743463+00:00"},{"alias_kind":"pith_short_8","alias_value":"WX2W2TUB","created_at":"2026-07-05T10:32:41.743463+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.17536","citing_title":"OmniDrive: An LLM-Choreographed Multi-Agent World Model with Unified Latent Co-Compression for Multi-View Driving Video Generation","ref_index":67,"is_internal_anchor":false},{"citing_arxiv_id":"2512.23994","citing_title":"PhyAVBench: A Challenging Audio Physics-Sensitivity Benchmark for Physically Grounded Text-to-Audio-Video Generation","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2512.23994","citing_title":"PhyAVBench: A Challenging Audio Physics-Sensitivity Benchmark for Physically Grounded Text-to-Audio-Video Generation","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2602.20981","citing_title":"Echoes Over Time: Unlocking Length Generalization in Video-to-Audio Generation Models","ref_index":50,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WX2W2TUBI3XNEX7HBTMWTHRRFM","json":"https://pith.science/pith/WX2W2TUBI3XNEX7HBTMWTHRRFM.json","graph_json":"https://pith.science/api/pith-number/WX2W2TUBI3XNEX7HBTMWTHRRFM/graph.json","events_json":"https://pith.science/api/pith-number/WX2W2TUBI3XNEX7HBTMWTHRRFM/events.json","paper":"https://pith.science/paper/WX2W2TUB"},"agent_actions":{"view_html":"https://pith.science/pith/WX2W2TUBI3XNEX7HBTMWTHRRFM","download_json":"https://pith.science/pith/WX2W2TUBI3XNEX7HBTMWTHRRFM.json","view_paper":"https://pith.science/paper/WX2W2TUB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.10719&json=true","fetch_graph":"https://pith.science/api/pith-number/WX2W2TUBI3XNEX7HBTMWTHRRFM/graph.json","fetch_events":"https://pith.science/api/pith-number/WX2W2TUBI3XNEX7HBTMWTHRRFM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WX2W2TUBI3XNEX7HBTMWTHRRFM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WX2W2TUBI3XNEX7HBTMWTHRRFM/action/storage_attestation","attest_author":"https://pith.science/pith/WX2W2TUBI3XNEX7HBTMWTHRRFM/action/author_attestation","sign_citation":"https://pith.science/pith/WX2W2TUBI3XNEX7HBTMWTHRRFM/action/citation_signature","submit_replication":"https://pith.science/pith/WX2W2TUBI3XNEX7HBTMWTHRRFM/action/replication_record"}},"created_at":"2026-07-05T10:32:41.743463+00:00","updated_at":"2026-07-05T10:32:41.743463+00:00"}