{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:TMSR36C5F7EOKDT6ECFWLKOKJ4","short_pith_number":"pith:TMSR36C5","schema_version":"1.0","canonical_sha256":"9b251df85d2fc8e50e7e208b65a9ca4f0aa4f89a0791e642ccac19b8781e4dd0","source":{"kind":"arxiv","id":"2011.03530","version":1},"attestation_state":"computed","paper":{"title":"Large-scale multilingual audio visual dubbing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CV","authors_text":"Brendan Shillingford, Eren Sezener, Luis C. Cobo, Miaosen Wang, Misha Denil, Nando de Freitas, Wendi Liu, Yannis Assael, Yi Yang, Yusuf Aytar, Yutian Chen, Yu Zhang","submitted_at":"2020-11-06T18:58:15Z","abstract_excerpt":"We describe a system for large-scale audiovisual translation and dubbing, which translates videos from one language to another. The source language's speech content is transcribed to text, translated, and automatically synthesized into target language speech using the original speaker's voice. The visual content is translated by synthesizing lip movements for the speaker to match the translated audio, creating a seamless audiovisual experience in the target language. The audio and visual translation subsystems each contain a large-scale generic synthesis model trained on thousands of hours of "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2011.03530","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2020-11-06T18:58:15Z","cross_cats_sorted":["cs.SD","eess.AS"],"title_canon_sha256":"f600e196a0e57f9b57d2c58640971e5ff003ff70d13a2ae3740b2e08cbcfa886","abstract_canon_sha256":"e7405400329e75b8fc25a7bd31ec9286a002e3ba817be2fe57f844bdc7d6be5b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:49:43.994931Z","signature_b64":"HSGnrUDSUClkOCxUx30Bdi3W1mJ7uFZzvAb5nN5IUAMR2O1uVVgzAjVUgHVMEE3/pBzYe+QxFfnt4G4JaLSvAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9b251df85d2fc8e50e7e208b65a9ca4f0aa4f89a0791e642ccac19b8781e4dd0","last_reissued_at":"2026-07-05T01:49:43.994398Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:49:43.994398Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Large-scale multilingual audio visual dubbing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CV","authors_text":"Brendan Shillingford, Eren Sezener, Luis C. Cobo, Miaosen Wang, Misha Denil, Nando de Freitas, Wendi Liu, Yannis Assael, Yi Yang, Yusuf Aytar, Yutian Chen, Yu Zhang","submitted_at":"2020-11-06T18:58:15Z","abstract_excerpt":"We describe a system for large-scale audiovisual translation and dubbing, which translates videos from one language to another. The source language's speech content is transcribed to text, translated, and automatically synthesized into target language speech using the original speaker's voice. The visual content is translated by synthesizing lip movements for the speaker to match the translated audio, creating a seamless audiovisual experience in the target language. The audio and visual translation subsystems each contain a large-scale generic synthesis model trained on thousands of hours of "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2011.03530","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2011.03530/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2011.03530","created_at":"2026-07-05T01:49:43.994468+00:00"},{"alias_kind":"arxiv_version","alias_value":"2011.03530v1","created_at":"2026-07-05T01:49:43.994468+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2011.03530","created_at":"2026-07-05T01:49:43.994468+00:00"},{"alias_kind":"pith_short_12","alias_value":"TMSR36C5F7EO","created_at":"2026-07-05T01:49:43.994468+00:00"},{"alias_kind":"pith_short_16","alias_value":"TMSR36C5F7EOKDT6","created_at":"2026-07-05T01:49:43.994468+00:00"},{"alias_kind":"pith_short_8","alias_value":"TMSR36C5","created_at":"2026-07-05T01:49:43.994468+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.09111","citing_title":"PS-TTS: Phonetic Synchronization in Text-to-Speech for Achieving Natural Automated Dubbing","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TMSR36C5F7EOKDT6ECFWLKOKJ4","json":"https://pith.science/pith/TMSR36C5F7EOKDT6ECFWLKOKJ4.json","graph_json":"https://pith.science/api/pith-number/TMSR36C5F7EOKDT6ECFWLKOKJ4/graph.json","events_json":"https://pith.science/api/pith-number/TMSR36C5F7EOKDT6ECFWLKOKJ4/events.json","paper":"https://pith.science/paper/TMSR36C5"},"agent_actions":{"view_html":"https://pith.science/pith/TMSR36C5F7EOKDT6ECFWLKOKJ4","download_json":"https://pith.science/pith/TMSR36C5F7EOKDT6ECFWLKOKJ4.json","view_paper":"https://pith.science/paper/TMSR36C5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2011.03530&json=true","fetch_graph":"https://pith.science/api/pith-number/TMSR36C5F7EOKDT6ECFWLKOKJ4/graph.json","fetch_events":"https://pith.science/api/pith-number/TMSR36C5F7EOKDT6ECFWLKOKJ4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TMSR36C5F7EOKDT6ECFWLKOKJ4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TMSR36C5F7EOKDT6ECFWLKOKJ4/action/storage_attestation","attest_author":"https://pith.science/pith/TMSR36C5F7EOKDT6ECFWLKOKJ4/action/author_attestation","sign_citation":"https://pith.science/pith/TMSR36C5F7EOKDT6ECFWLKOKJ4/action/citation_signature","submit_replication":"https://pith.science/pith/TMSR36C5F7EOKDT6ECFWLKOKJ4/action/replication_record"}},"created_at":"2026-07-05T01:49:43.994468+00:00","updated_at":"2026-07-05T01:49:43.994468+00:00"}