{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:U5IYKNGAXD4QBG4I2IAJH23OZL","short_pith_number":"pith:U5IYKNGA","schema_version":"1.0","canonical_sha256":"a7518534c0b8f9009b88d20093eb6ecad03452a1498f7a3277abdd1233445a65","source":{"kind":"arxiv","id":"2407.04652","version":1},"attestation_state":"computed","paper":{"title":"Pretraining End-to-End Keyword Search with Automatically Discovered Acoustic Units","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"eess.AS","authors_text":"Bolaji Yusuf, Jan \"Honza\" \\v{C}ernock\\'y, Murat Sara\\c{c}lar","submitted_at":"2024-07-05T17:07:58Z","abstract_excerpt":"End-to-end (E2E) keyword search (KWS) has emerged as an alternative and complimentary approach to conventional keyword search which depends on the output of automatic speech recognition (ASR) systems. While E2E methods greatly simplify the KWS pipeline, they generally have worse performance than their ASR-based counterparts, which can benefit from pretraining with untranscribed data. In this work, we propose a method for pretraining E2E KWS systems with untranscribed data, which involves using acoustic unit discovery (AUD) to obtain discrete units for untranscribed data and then learning to lo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.04652","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.AS","submitted_at":"2024-07-05T17:07:58Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"cd20436813b5c6da8bbab5cad4f94a255a0626d3a184d764fe1477d25846e0c4","abstract_canon_sha256":"fa02bc663418fffa83eaefdab718014d2df9a23b47a01e403f50c578588fe540"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:40:36.703204Z","signature_b64":"4PhrWA5+pruqIfmvOYFac682hP9xgSu4pSea3itVqgRJNRSRuhJg8qZAqimaoDFFPxSqKiYkrSksNI52A7O+CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a7518534c0b8f9009b88d20093eb6ecad03452a1498f7a3277abdd1233445a65","last_reissued_at":"2026-07-05T08:40:36.702643Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:40:36.702643Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Pretraining End-to-End Keyword Search with Automatically Discovered Acoustic Units","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"eess.AS","authors_text":"Bolaji Yusuf, Jan \"Honza\" \\v{C}ernock\\'y, Murat Sara\\c{c}lar","submitted_at":"2024-07-05T17:07:58Z","abstract_excerpt":"End-to-end (E2E) keyword search (KWS) has emerged as an alternative and complimentary approach to conventional keyword search which depends on the output of automatic speech recognition (ASR) systems. While E2E methods greatly simplify the KWS pipeline, they generally have worse performance than their ASR-based counterparts, which can benefit from pretraining with untranscribed data. In this work, we propose a method for pretraining E2E KWS systems with untranscribed data, which involves using acoustic unit discovery (AUD) to obtain discrete units for untranscribed data and then learning to lo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.04652","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.04652/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.04652","created_at":"2026-07-05T08:40:36.702704+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.04652v1","created_at":"2026-07-05T08:40:36.702704+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.04652","created_at":"2026-07-05T08:40:36.702704+00:00"},{"alias_kind":"pith_short_12","alias_value":"U5IYKNGAXD4Q","created_at":"2026-07-05T08:40:36.702704+00:00"},{"alias_kind":"pith_short_16","alias_value":"U5IYKNGAXD4QBG4I","created_at":"2026-07-05T08:40:36.702704+00:00"},{"alias_kind":"pith_short_8","alias_value":"U5IYKNGA","created_at":"2026-07-05T08:40:36.702704+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2501.03523","citing_title":"Vocal Tract Length Warped Features for Spoken Keyword Spotting","ref_index":19,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/U5IYKNGAXD4QBG4I2IAJH23OZL","json":"https://pith.science/pith/U5IYKNGAXD4QBG4I2IAJH23OZL.json","graph_json":"https://pith.science/api/pith-number/U5IYKNGAXD4QBG4I2IAJH23OZL/graph.json","events_json":"https://pith.science/api/pith-number/U5IYKNGAXD4QBG4I2IAJH23OZL/events.json","paper":"https://pith.science/paper/U5IYKNGA"},"agent_actions":{"view_html":"https://pith.science/pith/U5IYKNGAXD4QBG4I2IAJH23OZL","download_json":"https://pith.science/pith/U5IYKNGAXD4QBG4I2IAJH23OZL.json","view_paper":"https://pith.science/paper/U5IYKNGA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.04652&json=true","fetch_graph":"https://pith.science/api/pith-number/U5IYKNGAXD4QBG4I2IAJH23OZL/graph.json","fetch_events":"https://pith.science/api/pith-number/U5IYKNGAXD4QBG4I2IAJH23OZL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/U5IYKNGAXD4QBG4I2IAJH23OZL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/U5IYKNGAXD4QBG4I2IAJH23OZL/action/storage_attestation","attest_author":"https://pith.science/pith/U5IYKNGAXD4QBG4I2IAJH23OZL/action/author_attestation","sign_citation":"https://pith.science/pith/U5IYKNGAXD4QBG4I2IAJH23OZL/action/citation_signature","submit_replication":"https://pith.science/pith/U5IYKNGAXD4QBG4I2IAJH23OZL/action/replication_record"}},"created_at":"2026-07-05T08:40:36.702704+00:00","updated_at":"2026-07-05T08:40:36.702704+00:00"}