{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:2OZ6FKLRSRFOFBN52AOFDHOZVQ","short_pith_number":"pith:2OZ6FKLR","schema_version":"1.0","canonical_sha256":"d3b3e2a971944ae285bdd01c519dd9ac3128876bc27215a97bcd7ec5d2b3368f","source":{"kind":"arxiv","id":"2401.04235","version":1},"attestation_state":"computed","paper":{"title":"High-precision Voice Search Query Correction via Retrievable Speech-text Embedings","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Allen Chen, Andrew Rosenberg, Christopher Li, Diamantino Caseiro, Gary Wang, Heng Su, Kyle Kastner, Leonid Velikovich, Pat Rondon, Petar Aleksic, Zelin Wu, Zhehuai Chen","submitted_at":"2024-01-08T20:59:56Z","abstract_excerpt":"Automatic speech recognition (ASR) systems can suffer from poor recall for various reasons, such as noisy audio, lack of sufficient training data, etc.\n  Previous work has shown that recall can be improved by retrieving rewrite candidates from a large database of likely, contextually-relevant alternatives to the hypothesis text using nearest-neighbors search over embeddings of the ASR hypothesis text to correct and candidate corrections.\n  However, ASR-hypothesis-based retrieval can yield poor precision if the textual hypotheses are too phonetically dissimilar to the transcript truth. In this "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.04235","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-01-08T20:59:56Z","cross_cats_sorted":["cs.SD","eess.AS"],"title_canon_sha256":"397030b2e0c40236a7b4b8000944e8ecc47d77a652fbf303db1b1f82e095fdb0","abstract_canon_sha256":"23cbe87477ecad78a6b7b48e7fbf1881d463a24380edef6e2676c3bf3d811480"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:31:40.770971Z","signature_b64":"frOjptK9JPyMEGOBudeRyX8G49SwimM9ohB4T1otiEYfK6i79/m9jRTZIU6Z6iH6DQrRyI/eKlS9W8KfHJWuDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d3b3e2a971944ae285bdd01c519dd9ac3128876bc27215a97bcd7ec5d2b3368f","last_reissued_at":"2026-07-05T07:31:40.770482Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:31:40.770482Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"High-precision Voice Search Query Correction via Retrievable Speech-text Embedings","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Allen Chen, Andrew Rosenberg, Christopher Li, Diamantino Caseiro, Gary Wang, Heng Su, Kyle Kastner, Leonid Velikovich, Pat Rondon, Petar Aleksic, Zelin Wu, Zhehuai Chen","submitted_at":"2024-01-08T20:59:56Z","abstract_excerpt":"Automatic speech recognition (ASR) systems can suffer from poor recall for various reasons, such as noisy audio, lack of sufficient training data, etc.\n  Previous work has shown that recall can be improved by retrieving rewrite candidates from a large database of likely, contextually-relevant alternatives to the hypothesis text using nearest-neighbors search over embeddings of the ASR hypothesis text to correct and candidate corrections.\n  However, ASR-hypothesis-based retrieval can yield poor precision if the textual hypotheses are too phonetically dissimilar to the transcript truth. In this "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.04235","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.04235/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.04235","created_at":"2026-07-05T07:31:40.770540+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.04235v1","created_at":"2026-07-05T07:31:40.770540+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.04235","created_at":"2026-07-05T07:31:40.770540+00:00"},{"alias_kind":"pith_short_12","alias_value":"2OZ6FKLRSRFO","created_at":"2026-07-05T07:31:40.770540+00:00"},{"alias_kind":"pith_short_16","alias_value":"2OZ6FKLRSRFOFBN5","created_at":"2026-07-05T07:31:40.770540+00:00"},{"alias_kind":"pith_short_8","alias_value":"2OZ6FKLR","created_at":"2026-07-05T07:31:40.770540+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2508.07285","citing_title":"Non-Intrusive Automatic Speech Recognition Refinement: A Survey","ref_index":99,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2OZ6FKLRSRFOFBN52AOFDHOZVQ","json":"https://pith.science/pith/2OZ6FKLRSRFOFBN52AOFDHOZVQ.json","graph_json":"https://pith.science/api/pith-number/2OZ6FKLRSRFOFBN52AOFDHOZVQ/graph.json","events_json":"https://pith.science/api/pith-number/2OZ6FKLRSRFOFBN52AOFDHOZVQ/events.json","paper":"https://pith.science/paper/2OZ6FKLR"},"agent_actions":{"view_html":"https://pith.science/pith/2OZ6FKLRSRFOFBN52AOFDHOZVQ","download_json":"https://pith.science/pith/2OZ6FKLRSRFOFBN52AOFDHOZVQ.json","view_paper":"https://pith.science/paper/2OZ6FKLR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.04235&json=true","fetch_graph":"https://pith.science/api/pith-number/2OZ6FKLRSRFOFBN52AOFDHOZVQ/graph.json","fetch_events":"https://pith.science/api/pith-number/2OZ6FKLRSRFOFBN52AOFDHOZVQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2OZ6FKLRSRFOFBN52AOFDHOZVQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2OZ6FKLRSRFOFBN52AOFDHOZVQ/action/storage_attestation","attest_author":"https://pith.science/pith/2OZ6FKLRSRFOFBN52AOFDHOZVQ/action/author_attestation","sign_citation":"https://pith.science/pith/2OZ6FKLRSRFOFBN52AOFDHOZVQ/action/citation_signature","submit_replication":"https://pith.science/pith/2OZ6FKLRSRFOFBN52AOFDHOZVQ/action/replication_record"}},"created_at":"2026-07-05T07:31:40.770540+00:00","updated_at":"2026-07-05T07:31:40.770540+00:00"}