{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:RMZ6TIGV3TQTCA3XKAOFS47QWR","short_pith_number":"pith:RMZ6TIGV","schema_version":"1.0","canonical_sha256":"8b33e9a0d5dce1310377501c5973f0b44c829c64f6c8844fc97bec8a0fc55b1d","source":{"kind":"arxiv","id":"2104.04487","version":1},"attestation_state":"computed","paper":{"title":"Language model fusion for streaming end to end speech recognition","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG","cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Anjuli Kannan, Eugene Weinstein, MohammadReza Ghodsi, Rodrigo Cabrera, Xiaofeng Liu, Zebulun Matteson","submitted_at":"2021-04-09T17:14:28Z","abstract_excerpt":"Streaming processing of speech audio is required for many contemporary practical speech recognition tasks. Even with the large corpora of manually transcribed speech data available today, it is impossible for such corpora to cover adequately the long tail of linguistic content that's important for tasks such as open-ended dictation and voice search. We seek to address both the streaming and the tail recognition challenges by using a language model (LM) trained on unpaired text data to enhance the end-to-end (E2E) model. We extend shallow fusion and cold fusion approaches to streaming Recurrent"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2104.04487","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2021-04-09T17:14:28Z","cross_cats_sorted":["cs.LG","cs.SD","eess.AS"],"title_canon_sha256":"df4353553085908e391c3a70175f3a43c3655d87dedb510030a33176a9855d1b","abstract_canon_sha256":"d171d161f7eda5f28f2b58deeca12de90ca14d891c9d0f65109c377a76086e39"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:30:39.663772Z","signature_b64":"FcnWK/Tzdg1gEs9U5sRQQigv8hYB1XT8iVd9YA96wB6YiSB4zW73Vxdi4KcEA2h7/+8Nh8T1dz7gV6O1GOZyDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8b33e9a0d5dce1310377501c5973f0b44c829c64f6c8844fc97bec8a0fc55b1d","last_reissued_at":"2026-07-05T02:30:39.663341Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:30:39.663341Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Language model fusion for streaming end to end speech recognition","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG","cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Anjuli Kannan, Eugene Weinstein, MohammadReza Ghodsi, Rodrigo Cabrera, Xiaofeng Liu, Zebulun Matteson","submitted_at":"2021-04-09T17:14:28Z","abstract_excerpt":"Streaming processing of speech audio is required for many contemporary practical speech recognition tasks. Even with the large corpora of manually transcribed speech data available today, it is impossible for such corpora to cover adequately the long tail of linguistic content that's important for tasks such as open-ended dictation and voice search. We seek to address both the streaming and the tail recognition challenges by using a language model (LM) trained on unpaired text data to enhance the end-to-end (E2E) model. We extend shallow fusion and cold fusion approaches to streaming Recurrent"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2104.04487","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2104.04487/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2104.04487","created_at":"2026-07-05T02:30:39.663403+00:00"},{"alias_kind":"arxiv_version","alias_value":"2104.04487v1","created_at":"2026-07-05T02:30:39.663403+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2104.04487","created_at":"2026-07-05T02:30:39.663403+00:00"},{"alias_kind":"pith_short_12","alias_value":"RMZ6TIGV3TQT","created_at":"2026-07-05T02:30:39.663403+00:00"},{"alias_kind":"pith_short_16","alias_value":"RMZ6TIGV3TQTCA3X","created_at":"2026-07-05T02:30:39.663403+00:00"},{"alias_kind":"pith_short_8","alias_value":"RMZ6TIGV","created_at":"2026-07-05T02:30:39.663403+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RMZ6TIGV3TQTCA3XKAOFS47QWR","json":"https://pith.science/pith/RMZ6TIGV3TQTCA3XKAOFS47QWR.json","graph_json":"https://pith.science/api/pith-number/RMZ6TIGV3TQTCA3XKAOFS47QWR/graph.json","events_json":"https://pith.science/api/pith-number/RMZ6TIGV3TQTCA3XKAOFS47QWR/events.json","paper":"https://pith.science/paper/RMZ6TIGV"},"agent_actions":{"view_html":"https://pith.science/pith/RMZ6TIGV3TQTCA3XKAOFS47QWR","download_json":"https://pith.science/pith/RMZ6TIGV3TQTCA3XKAOFS47QWR.json","view_paper":"https://pith.science/paper/RMZ6TIGV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2104.04487&json=true","fetch_graph":"https://pith.science/api/pith-number/RMZ6TIGV3TQTCA3XKAOFS47QWR/graph.json","fetch_events":"https://pith.science/api/pith-number/RMZ6TIGV3TQTCA3XKAOFS47QWR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RMZ6TIGV3TQTCA3XKAOFS47QWR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RMZ6TIGV3TQTCA3XKAOFS47QWR/action/storage_attestation","attest_author":"https://pith.science/pith/RMZ6TIGV3TQTCA3XKAOFS47QWR/action/author_attestation","sign_citation":"https://pith.science/pith/RMZ6TIGV3TQTCA3XKAOFS47QWR/action/citation_signature","submit_replication":"https://pith.science/pith/RMZ6TIGV3TQTCA3XKAOFS47QWR/action/replication_record"}},"created_at":"2026-07-05T02:30:39.663403+00:00","updated_at":"2026-07-05T02:30:39.663403+00:00"}