{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:LBATHC6EJUDCLT3E6ANQ2GVQX2","short_pith_number":"pith:LBATHC6E","schema_version":"1.0","canonical_sha256":"5841338bc44d0625cf64f01b0d1ab0bea11f1ee8e9ad89d3a5d66845186220f0","source":{"kind":"arxiv","id":"2408.11593","version":3},"attestation_state":"computed","paper":{"title":"MCDubber: Multimodal Context-Aware Expressive Video Dubbing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","cs.SD","eess.AS"],"primary_cat":"cs.MM","authors_text":"De Hu, Feilong Bao, Guanglai Gao, Rui Liu, Yuan Zhao, Zhenqi Jia","submitted_at":"2024-08-21T12:59:42Z","abstract_excerpt":"Automatic Video Dubbing (AVD) aims to take the given script and generate speech that aligns with lip motion and prosody expressiveness. Current AVD models mainly utilize visual information of the current sentence to enhance the prosody of synthesized speech. However, it is crucial to consider whether the prosody of the generated dubbing aligns with the multimodal context, as the dubbing will be combined with the original context in the final video. This aspect has been overlooked in previous studies. To address this issue, we propose a Multimodal Context-aware video Dubbing model, termed \\text"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.11593","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.MM","submitted_at":"2024-08-21T12:59:42Z","cross_cats_sorted":["cs.CV","cs.SD","eess.AS"],"title_canon_sha256":"9591cabd0ea7f5a9e583b706c7d756e403918b8f53007fb668af9515f06101e7","abstract_canon_sha256":"6bcc22a1c30fc67dac230374194c678b42e511d7b3c9ba72dd89f8e02701167a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:02:41.763877Z","signature_b64":"iY+kdZuBxrKRiVLIU0jN+6HeP8NSsVYJILEcmnXhSFyQh86FDZXXCGq7Ux+pDPWwoPjRkmuqfbqHu52nNUQyAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5841338bc44d0625cf64f01b0d1ab0bea11f1ee8e9ad89d3a5d66845186220f0","last_reissued_at":"2026-07-05T09:02:41.763361Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:02:41.763361Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MCDubber: Multimodal Context-Aware Expressive Video Dubbing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","cs.SD","eess.AS"],"primary_cat":"cs.MM","authors_text":"De Hu, Feilong Bao, Guanglai Gao, Rui Liu, Yuan Zhao, Zhenqi Jia","submitted_at":"2024-08-21T12:59:42Z","abstract_excerpt":"Automatic Video Dubbing (AVD) aims to take the given script and generate speech that aligns with lip motion and prosody expressiveness. Current AVD models mainly utilize visual information of the current sentence to enhance the prosody of synthesized speech. However, it is crucial to consider whether the prosody of the generated dubbing aligns with the multimodal context, as the dubbing will be combined with the original context in the final video. This aspect has been overlooked in previous studies. To address this issue, we propose a Multimodal Context-aware video Dubbing model, termed \\text"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.11593","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.11593/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.11593","created_at":"2026-07-05T09:02:41.763423+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.11593v3","created_at":"2026-07-05T09:02:41.763423+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.11593","created_at":"2026-07-05T09:02:41.763423+00:00"},{"alias_kind":"pith_short_12","alias_value":"LBATHC6EJUDC","created_at":"2026-07-05T09:02:41.763423+00:00"},{"alias_kind":"pith_short_16","alias_value":"LBATHC6EJUDCLT3E","created_at":"2026-07-05T09:02:41.763423+00:00"},{"alias_kind":"pith_short_8","alias_value":"LBATHC6E","created_at":"2026-07-05T09:02:41.763423+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.12292","citing_title":"CoSyncDiT: Cognitive Synchronous Diffusion Transformer for Movie Dubbing","ref_index":53,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LBATHC6EJUDCLT3E6ANQ2GVQX2","json":"https://pith.science/pith/LBATHC6EJUDCLT3E6ANQ2GVQX2.json","graph_json":"https://pith.science/api/pith-number/LBATHC6EJUDCLT3E6ANQ2GVQX2/graph.json","events_json":"https://pith.science/api/pith-number/LBATHC6EJUDCLT3E6ANQ2GVQX2/events.json","paper":"https://pith.science/paper/LBATHC6E"},"agent_actions":{"view_html":"https://pith.science/pith/LBATHC6EJUDCLT3E6ANQ2GVQX2","download_json":"https://pith.science/pith/LBATHC6EJUDCLT3E6ANQ2GVQX2.json","view_paper":"https://pith.science/paper/LBATHC6E","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.11593&json=true","fetch_graph":"https://pith.science/api/pith-number/LBATHC6EJUDCLT3E6ANQ2GVQX2/graph.json","fetch_events":"https://pith.science/api/pith-number/LBATHC6EJUDCLT3E6ANQ2GVQX2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LBATHC6EJUDCLT3E6ANQ2GVQX2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LBATHC6EJUDCLT3E6ANQ2GVQX2/action/storage_attestation","attest_author":"https://pith.science/pith/LBATHC6EJUDCLT3E6ANQ2GVQX2/action/author_attestation","sign_citation":"https://pith.science/pith/LBATHC6EJUDCLT3E6ANQ2GVQX2/action/citation_signature","submit_replication":"https://pith.science/pith/LBATHC6EJUDCLT3E6ANQ2GVQX2/action/replication_record"}},"created_at":"2026-07-05T09:02:41.763423+00:00","updated_at":"2026-07-05T09:02:41.763423+00:00"}