{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:ENBP2GALL4PD6E4GXHA3JHARK5","short_pith_number":"pith:ENBP2GAL","schema_version":"1.0","canonical_sha256":"2342fd180b5f1e3f1386b9c1b49c11576cf01549f493b45049820fc5edf8c68f","source":{"kind":"arxiv","id":"2106.02896","version":1},"attestation_state":"computed","paper":{"title":"Human Listening and Live Captioning: Multi-Task Training for Speech Enhancement","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"eess.AS","authors_text":"Hemin Yang, Huaming Wang, Min Tang, Sefik Emre Eskimez, Takuya Yoshioka, Xiaofei Wang, Zhuo Chen, Zirun Zhu","submitted_at":"2021-06-05T13:40:53Z","abstract_excerpt":"With the surge of online meetings, it has become more critical than ever to provide high-quality speech audio and live captioning under various noise conditions. However, most monaural speech enhancement (SE) models introduce processing artifacts and thus degrade the performance of downstream tasks, including automatic speech recognition (ASR). This paper proposes a multi-task training framework to make the SE models unharmful to ASR. Because most ASR training samples do not have corresponding clean signal references, we alternately perform two model update steps called SE-step and ASR-step. T"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2106.02896","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.AS","submitted_at":"2021-06-05T13:40:53Z","cross_cats_sorted":[],"title_canon_sha256":"fdb63fd6ce3a75bbe0b1a946a106058a03f70976b28e835ed3148bd5b8b47375","abstract_canon_sha256":"da1f32e0af776c6fd55d7ff7d49b3c522009722ec15d04348c47d595b40e8665"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:46:31.585170Z","signature_b64":"T4mWClWBA57Sh7fRn1Loto2E5nRQctQ0lvek4Lf/v1mSAAwHamwYwmZP7mJp0MHWVOR+MV6IS9N4D13C8tdMAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2342fd180b5f1e3f1386b9c1b49c11576cf01549f493b45049820fc5edf8c68f","last_reissued_at":"2026-07-05T02:46:31.584750Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:46:31.584750Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Human Listening and Live Captioning: Multi-Task Training for Speech Enhancement","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"eess.AS","authors_text":"Hemin Yang, Huaming Wang, Min Tang, Sefik Emre Eskimez, Takuya Yoshioka, Xiaofei Wang, Zhuo Chen, Zirun Zhu","submitted_at":"2021-06-05T13:40:53Z","abstract_excerpt":"With the surge of online meetings, it has become more critical than ever to provide high-quality speech audio and live captioning under various noise conditions. However, most monaural speech enhancement (SE) models introduce processing artifacts and thus degrade the performance of downstream tasks, including automatic speech recognition (ASR). This paper proposes a multi-task training framework to make the SE models unharmful to ASR. Because most ASR training samples do not have corresponding clean signal references, we alternately perform two model update steps called SE-step and ASR-step. T"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2106.02896","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2106.02896/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2106.02896","created_at":"2026-07-05T02:46:31.584810+00:00"},{"alias_kind":"arxiv_version","alias_value":"2106.02896v1","created_at":"2026-07-05T02:46:31.584810+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2106.02896","created_at":"2026-07-05T02:46:31.584810+00:00"},{"alias_kind":"pith_short_12","alias_value":"ENBP2GALL4PD","created_at":"2026-07-05T02:46:31.584810+00:00"},{"alias_kind":"pith_short_16","alias_value":"ENBP2GALL4PD6E4G","created_at":"2026-07-05T02:46:31.584810+00:00"},{"alias_kind":"pith_short_8","alias_value":"ENBP2GAL","created_at":"2026-07-05T02:46:31.584810+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ENBP2GALL4PD6E4GXHA3JHARK5","json":"https://pith.science/pith/ENBP2GALL4PD6E4GXHA3JHARK5.json","graph_json":"https://pith.science/api/pith-number/ENBP2GALL4PD6E4GXHA3JHARK5/graph.json","events_json":"https://pith.science/api/pith-number/ENBP2GALL4PD6E4GXHA3JHARK5/events.json","paper":"https://pith.science/paper/ENBP2GAL"},"agent_actions":{"view_html":"https://pith.science/pith/ENBP2GALL4PD6E4GXHA3JHARK5","download_json":"https://pith.science/pith/ENBP2GALL4PD6E4GXHA3JHARK5.json","view_paper":"https://pith.science/paper/ENBP2GAL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2106.02896&json=true","fetch_graph":"https://pith.science/api/pith-number/ENBP2GALL4PD6E4GXHA3JHARK5/graph.json","fetch_events":"https://pith.science/api/pith-number/ENBP2GALL4PD6E4GXHA3JHARK5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ENBP2GALL4PD6E4GXHA3JHARK5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ENBP2GALL4PD6E4GXHA3JHARK5/action/storage_attestation","attest_author":"https://pith.science/pith/ENBP2GALL4PD6E4GXHA3JHARK5/action/author_attestation","sign_citation":"https://pith.science/pith/ENBP2GALL4PD6E4GXHA3JHARK5/action/citation_signature","submit_replication":"https://pith.science/pith/ENBP2GALL4PD6E4GXHA3JHARK5/action/replication_record"}},"created_at":"2026-07-05T02:46:31.584810+00:00","updated_at":"2026-07-05T02:46:31.584810+00:00"}