{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:W4MIQZLRV2HM7A4NH5MQFF7Q7X","short_pith_number":"pith:W4MIQZLR","schema_version":"1.0","canonical_sha256":"b718886571ae8ecf838d3f590297f0fdf24000f005d54aaaad87973bb858d50e","source":{"kind":"arxiv","id":"2010.04556","version":2},"attestation_state":"computed","paper":{"title":"Audio-Visual Speech Inpainting with Deep Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","eess.IV"],"primary_cat":"eess.AS","authors_text":"Daniel Michelsanti, Giovanni Morrone, Jesper Jensen, Zheng-Hua Tan","submitted_at":"2020-10-09T13:23:01Z","abstract_excerpt":"In this paper, we present a deep-learning-based framework for audio-visual speech inpainting, i.e., the task of restoring the missing parts of an acoustic speech signal from reliable audio context and uncorrupted visual information. Recent work focuses solely on audio-only methods and generally aims at inpainting music signals, which show highly different structure than speech. Instead, we inpaint speech signals with gaps ranging from 100 ms to 1600 ms to investigate the contribution that vision can provide for gaps of different duration. We also experiment with a multi-task learning approach "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2010.04556","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.AS","submitted_at":"2020-10-09T13:23:01Z","cross_cats_sorted":["cs.LG","eess.IV"],"title_canon_sha256":"3f17af5d84e77caa4e9f6fb33eeb9fad8e452482525e6a5b8d59d64e1193621f","abstract_canon_sha256":"3b5241d2d8534b48085ba9ec1182785e9adc09266beee0b300cec85eaba77311"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:12:30.831135Z","signature_b64":"GGuNeiaOzO08geOfuGlPsadwC4Te6i56XnYRSELK9DmHNkcN4P9O+VNK2ccEG40OJjqoIOepYEE3fvkGPoZbCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b718886571ae8ecf838d3f590297f0fdf24000f005d54aaaad87973bb858d50e","last_reissued_at":"2026-07-05T02:12:30.830714Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:12:30.830714Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Audio-Visual Speech Inpainting with Deep Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","eess.IV"],"primary_cat":"eess.AS","authors_text":"Daniel Michelsanti, Giovanni Morrone, Jesper Jensen, Zheng-Hua Tan","submitted_at":"2020-10-09T13:23:01Z","abstract_excerpt":"In this paper, we present a deep-learning-based framework for audio-visual speech inpainting, i.e., the task of restoring the missing parts of an acoustic speech signal from reliable audio context and uncorrupted visual information. Recent work focuses solely on audio-only methods and generally aims at inpainting music signals, which show highly different structure than speech. Instead, we inpaint speech signals with gaps ranging from 100 ms to 1600 ms to investigate the contribution that vision can provide for gaps of different duration. We also experiment with a multi-task learning approach "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2010.04556","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2010.04556/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2010.04556","created_at":"2026-07-05T02:12:30.830772+00:00"},{"alias_kind":"arxiv_version","alias_value":"2010.04556v2","created_at":"2026-07-05T02:12:30.830772+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2010.04556","created_at":"2026-07-05T02:12:30.830772+00:00"},{"alias_kind":"pith_short_12","alias_value":"W4MIQZLRV2HM","created_at":"2026-07-05T02:12:30.830772+00:00"},{"alias_kind":"pith_short_16","alias_value":"W4MIQZLRV2HM7A4N","created_at":"2026-07-05T02:12:30.830772+00:00"},{"alias_kind":"pith_short_8","alias_value":"W4MIQZLR","created_at":"2026-07-05T02:12:30.830772+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.21153","citing_title":"WaveLLDM: Design and Development of a Lightweight Latent Diffusion Model for Speech Enhancement and Restoration","ref_index":35,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/W4MIQZLRV2HM7A4NH5MQFF7Q7X","json":"https://pith.science/pith/W4MIQZLRV2HM7A4NH5MQFF7Q7X.json","graph_json":"https://pith.science/api/pith-number/W4MIQZLRV2HM7A4NH5MQFF7Q7X/graph.json","events_json":"https://pith.science/api/pith-number/W4MIQZLRV2HM7A4NH5MQFF7Q7X/events.json","paper":"https://pith.science/paper/W4MIQZLR"},"agent_actions":{"view_html":"https://pith.science/pith/W4MIQZLRV2HM7A4NH5MQFF7Q7X","download_json":"https://pith.science/pith/W4MIQZLRV2HM7A4NH5MQFF7Q7X.json","view_paper":"https://pith.science/paper/W4MIQZLR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2010.04556&json=true","fetch_graph":"https://pith.science/api/pith-number/W4MIQZLRV2HM7A4NH5MQFF7Q7X/graph.json","fetch_events":"https://pith.science/api/pith-number/W4MIQZLRV2HM7A4NH5MQFF7Q7X/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/W4MIQZLRV2HM7A4NH5MQFF7Q7X/action/timestamp_anchor","attest_storage":"https://pith.science/pith/W4MIQZLRV2HM7A4NH5MQFF7Q7X/action/storage_attestation","attest_author":"https://pith.science/pith/W4MIQZLRV2HM7A4NH5MQFF7Q7X/action/author_attestation","sign_citation":"https://pith.science/pith/W4MIQZLRV2HM7A4NH5MQFF7Q7X/action/citation_signature","submit_replication":"https://pith.science/pith/W4MIQZLRV2HM7A4NH5MQFF7Q7X/action/replication_record"}},"created_at":"2026-07-05T02:12:30.830772+00:00","updated_at":"2026-07-05T02:12:30.830772+00:00"}