{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:I4W6BNSBA5ZMQHYANE2ZNM655O","short_pith_number":"pith:I4W6BNSB","schema_version":"1.0","canonical_sha256":"472de0b6410772c81f00693596b3ddebbe5946ec6dab0749b93cdd8362fc054b","source":{"kind":"arxiv","id":"2212.11377","version":1},"attestation_state":"computed","paper":{"title":"ReVISE: Self-Supervised Speech Resynthesis with Visual Input for Universal and Generalized Speech Enhancement","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","cs.LG","cs.SD"],"primary_cat":"eess.AS","authors_text":"Bowen Shi, Jacob Donley, Tal Remez, Wei-Ning Hsu, Yossi Adi","submitted_at":"2022-12-21T21:36:52Z","abstract_excerpt":"Prior works on improving speech quality with visual input typically study each type of auditory distortion separately (e.g., separation, inpainting, video-to-speech) and present tailored algorithms. This paper proposes to unify these subjects and study Generalized Speech Enhancement, where the goal is not to reconstruct the exact reference clean signal, but to focus on improving certain aspects of speech. In particular, this paper concerns intelligibility, quality, and video synchronization. We cast the problem as audio-visual speech resynthesis, which is composed of two steps: pseudo audio-vi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2212.11377","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.AS","submitted_at":"2022-12-21T21:36:52Z","cross_cats_sorted":["cs.CV","cs.LG","cs.SD"],"title_canon_sha256":"e9d85e1c65bad9ba8cd2a403b7c734e16fbcddc3aba109491e64eb3c9b8f409f","abstract_canon_sha256":"47e877341767e763a679b166f9908adbc2ab1d64ba00992d8d2aa8bc3c5a0df9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:27:38.346208Z","signature_b64":"cdW60Ntoqr47hceSpA6v63d6TVnhuNx8TOn6EFhgJ6KI/jiZt9ZSD/bJyI9lzhGeMxYTX2gdtoawOS+8jbe4Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"472de0b6410772c81f00693596b3ddebbe5946ec6dab0749b93cdd8362fc054b","last_reissued_at":"2026-07-05T05:27:38.345785Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:27:38.345785Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ReVISE: Self-Supervised Speech Resynthesis with Visual Input for Universal and Generalized Speech Enhancement","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","cs.LG","cs.SD"],"primary_cat":"eess.AS","authors_text":"Bowen Shi, Jacob Donley, Tal Remez, Wei-Ning Hsu, Yossi Adi","submitted_at":"2022-12-21T21:36:52Z","abstract_excerpt":"Prior works on improving speech quality with visual input typically study each type of auditory distortion separately (e.g., separation, inpainting, video-to-speech) and present tailored algorithms. This paper proposes to unify these subjects and study Generalized Speech Enhancement, where the goal is not to reconstruct the exact reference clean signal, but to focus on improving certain aspects of speech. In particular, this paper concerns intelligibility, quality, and video synchronization. We cast the problem as audio-visual speech resynthesis, which is composed of two steps: pseudo audio-vi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2212.11377","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2212.11377/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2212.11377","created_at":"2026-07-05T05:27:38.345848+00:00"},{"alias_kind":"arxiv_version","alias_value":"2212.11377v1","created_at":"2026-07-05T05:27:38.345848+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2212.11377","created_at":"2026-07-05T05:27:38.345848+00:00"},{"alias_kind":"pith_short_12","alias_value":"I4W6BNSBA5ZM","created_at":"2026-07-05T05:27:38.345848+00:00"},{"alias_kind":"pith_short_16","alias_value":"I4W6BNSBA5ZMQHYA","created_at":"2026-07-05T05:27:38.345848+00:00"},{"alias_kind":"pith_short_8","alias_value":"I4W6BNSB","created_at":"2026-07-05T05:27:38.345848+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.02942","citing_title":"GenSE: Generative Speech Enhancement via Language Models using Hierarchical Modeling","ref_index":13,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/I4W6BNSBA5ZMQHYANE2ZNM655O","json":"https://pith.science/pith/I4W6BNSBA5ZMQHYANE2ZNM655O.json","graph_json":"https://pith.science/api/pith-number/I4W6BNSBA5ZMQHYANE2ZNM655O/graph.json","events_json":"https://pith.science/api/pith-number/I4W6BNSBA5ZMQHYANE2ZNM655O/events.json","paper":"https://pith.science/paper/I4W6BNSB"},"agent_actions":{"view_html":"https://pith.science/pith/I4W6BNSBA5ZMQHYANE2ZNM655O","download_json":"https://pith.science/pith/I4W6BNSBA5ZMQHYANE2ZNM655O.json","view_paper":"https://pith.science/paper/I4W6BNSB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2212.11377&json=true","fetch_graph":"https://pith.science/api/pith-number/I4W6BNSBA5ZMQHYANE2ZNM655O/graph.json","fetch_events":"https://pith.science/api/pith-number/I4W6BNSBA5ZMQHYANE2ZNM655O/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/I4W6BNSBA5ZMQHYANE2ZNM655O/action/timestamp_anchor","attest_storage":"https://pith.science/pith/I4W6BNSBA5ZMQHYANE2ZNM655O/action/storage_attestation","attest_author":"https://pith.science/pith/I4W6BNSBA5ZMQHYANE2ZNM655O/action/author_attestation","sign_citation":"https://pith.science/pith/I4W6BNSBA5ZMQHYANE2ZNM655O/action/citation_signature","submit_replication":"https://pith.science/pith/I4W6BNSBA5ZMQHYANE2ZNM655O/action/replication_record"}},"created_at":"2026-07-05T05:27:38.345848+00:00","updated_at":"2026-07-05T05:27:38.345848+00:00"}