{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:3M654QHAYZVDR4Q5B4RLGTB2VK","short_pith_number":"pith:3M654QHA","schema_version":"1.0","canonical_sha256":"db3dde40e0c66a38f21d0f22b34c3aaa8c1d39e3b7cd9036ef5db9b0c94eed23","source":{"kind":"arxiv","id":"2011.10538","version":3},"attestation_state":"computed","paper":{"title":"Improving RNN-T ASR Accuracy Using Context Audio","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD"],"primary_cat":"eess.AS","authors_text":"Andreas Schwarz, Ilya Sklyar, Simon Wiesler","submitted_at":"2020-11-20T18:16:04Z","abstract_excerpt":"We present a training scheme for streaming automatic speech recognition (ASR) based on recurrent neural network transducers (RNN-T) which allows the encoder network to learn to exploit context audio from a stream, using segmented or partially labeled sequences of the stream during training. We show that the use of context audio during training and inference can lead to word error rate reductions of more than 6% in a realistic production setting for a voice assistant ASR system. We investigate the effect of the proposed training approach on acoustically challenging data containing background sp"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2011.10538","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.AS","submitted_at":"2020-11-20T18:16:04Z","cross_cats_sorted":["cs.SD"],"title_canon_sha256":"fb8896ff3147b4d7d6dde99d33d7d78fe62d255d6e788aff6fb7b615d0b8111e","abstract_canon_sha256":"306c14a357a009f3a5219a7171b6236b3f3c10563bf6a3790134ee3320d48102"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:49:21.042552Z","signature_b64":"VQGZngZohhVZGlVMVMMmm/R3ML9jnH+ND0ltA2mjccHgv99j/FmM4H1PI0djgMovMRAp7cm5T99KAick0OevDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"db3dde40e0c66a38f21d0f22b34c3aaa8c1d39e3b7cd9036ef5db9b0c94eed23","last_reissued_at":"2026-07-05T02:49:21.042068Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:49:21.042068Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Improving RNN-T ASR Accuracy Using Context Audio","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD"],"primary_cat":"eess.AS","authors_text":"Andreas Schwarz, Ilya Sklyar, Simon Wiesler","submitted_at":"2020-11-20T18:16:04Z","abstract_excerpt":"We present a training scheme for streaming automatic speech recognition (ASR) based on recurrent neural network transducers (RNN-T) which allows the encoder network to learn to exploit context audio from a stream, using segmented or partially labeled sequences of the stream during training. We show that the use of context audio during training and inference can lead to word error rate reductions of more than 6% in a realistic production setting for a voice assistant ASR system. We investigate the effect of the proposed training approach on acoustically challenging data containing background sp"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2011.10538","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2011.10538/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2011.10538","created_at":"2026-07-05T02:49:21.042123+00:00"},{"alias_kind":"arxiv_version","alias_value":"2011.10538v3","created_at":"2026-07-05T02:49:21.042123+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2011.10538","created_at":"2026-07-05T02:49:21.042123+00:00"},{"alias_kind":"pith_short_12","alias_value":"3M654QHAYZVD","created_at":"2026-07-05T02:49:21.042123+00:00"},{"alias_kind":"pith_short_16","alias_value":"3M654QHAYZVDR4Q5","created_at":"2026-07-05T02:49:21.042123+00:00"},{"alias_kind":"pith_short_8","alias_value":"3M654QHA","created_at":"2026-07-05T02:49:21.042123+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.10456","citing_title":"Exploring Cross-Utterance Speech Contexts for Conformer-Transducer Speech Recognition Systems","ref_index":61,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3M654QHAYZVDR4Q5B4RLGTB2VK","json":"https://pith.science/pith/3M654QHAYZVDR4Q5B4RLGTB2VK.json","graph_json":"https://pith.science/api/pith-number/3M654QHAYZVDR4Q5B4RLGTB2VK/graph.json","events_json":"https://pith.science/api/pith-number/3M654QHAYZVDR4Q5B4RLGTB2VK/events.json","paper":"https://pith.science/paper/3M654QHA"},"agent_actions":{"view_html":"https://pith.science/pith/3M654QHAYZVDR4Q5B4RLGTB2VK","download_json":"https://pith.science/pith/3M654QHAYZVDR4Q5B4RLGTB2VK.json","view_paper":"https://pith.science/paper/3M654QHA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2011.10538&json=true","fetch_graph":"https://pith.science/api/pith-number/3M654QHAYZVDR4Q5B4RLGTB2VK/graph.json","fetch_events":"https://pith.science/api/pith-number/3M654QHAYZVDR4Q5B4RLGTB2VK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3M654QHAYZVDR4Q5B4RLGTB2VK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3M654QHAYZVDR4Q5B4RLGTB2VK/action/storage_attestation","attest_author":"https://pith.science/pith/3M654QHAYZVDR4Q5B4RLGTB2VK/action/author_attestation","sign_citation":"https://pith.science/pith/3M654QHAYZVDR4Q5B4RLGTB2VK/action/citation_signature","submit_replication":"https://pith.science/pith/3M654QHAYZVDR4Q5B4RLGTB2VK/action/replication_record"}},"created_at":"2026-07-05T02:49:21.042123+00:00","updated_at":"2026-07-05T02:49:21.042123+00:00"}