{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:UZ5I466QO7ZFSEVU4HWHGP5CAP","short_pith_number":"pith:UZ5I466Q","schema_version":"1.0","canonical_sha256":"a67a8e7bd077f25912b4e1ec733fa203df1f9c625331bf446ab638c1da49f2c8","source":{"kind":"arxiv","id":"2204.05352","version":2},"attestation_state":"computed","paper":{"title":"Large-Scale Streaming End-to-End Speech Translation with Neural Transducers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["eess.AS"],"primary_cat":"cs.CL","authors_text":"Jian Xue, Jinyu Li, Matt Post, Peidong Wang, Yashesh Gaur","submitted_at":"2022-04-11T18:18:53Z","abstract_excerpt":"Neural transducers have been widely used in automatic speech recognition (ASR). In this paper, we introduce it to streaming end-to-end speech translation (ST), which aims to convert audio signals to texts in other languages directly. Compared with cascaded ST that performs ASR followed by text-based machine translation (MT), the proposed Transformer transducer (TT)-based ST model drastically reduces inference latency, exploits speech information, and avoids error propagation from ASR to MT. To improve the modeling capacity, we propose attention pooling for the joint network in TT. In addition,"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2204.05352","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2022-04-11T18:18:53Z","cross_cats_sorted":["eess.AS"],"title_canon_sha256":"2a4e24ee55b0e11ff18b3a414e9959a6e213e3ba5441006fd7f32dfcce5e226b","abstract_canon_sha256":"b47cdee0e493e06bda25c90b6cc2a042606aee16469908c89ff181eda93436ad"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:36:48.432516Z","signature_b64":"pUprG0zLiYVOUuQ5oDfS9nbcOS1oNpG03NV+p8BzUSKW2javQ+mwmUCO+aaO4tmRn6ZZLc2wz47xaYq2qgpUCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a67a8e7bd077f25912b4e1ec733fa203df1f9c625331bf446ab638c1da49f2c8","last_reissued_at":"2026-07-05T04:36:48.431912Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:36:48.431912Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Large-Scale Streaming End-to-End Speech Translation with Neural Transducers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["eess.AS"],"primary_cat":"cs.CL","authors_text":"Jian Xue, Jinyu Li, Matt Post, Peidong Wang, Yashesh Gaur","submitted_at":"2022-04-11T18:18:53Z","abstract_excerpt":"Neural transducers have been widely used in automatic speech recognition (ASR). In this paper, we introduce it to streaming end-to-end speech translation (ST), which aims to convert audio signals to texts in other languages directly. Compared with cascaded ST that performs ASR followed by text-based machine translation (MT), the proposed Transformer transducer (TT)-based ST model drastically reduces inference latency, exploits speech information, and avoids error propagation from ASR to MT. To improve the modeling capacity, we propose attention pooling for the joint network in TT. In addition,"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2204.05352","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2204.05352/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2204.05352","created_at":"2026-07-05T04:36:48.431977+00:00"},{"alias_kind":"arxiv_version","alias_value":"2204.05352v2","created_at":"2026-07-05T04:36:48.431977+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2204.05352","created_at":"2026-07-05T04:36:48.431977+00:00"},{"alias_kind":"pith_short_12","alias_value":"UZ5I466QO7ZF","created_at":"2026-07-05T04:36:48.431977+00:00"},{"alias_kind":"pith_short_16","alias_value":"UZ5I466QO7ZFSEVU","created_at":"2026-07-05T04:36:48.431977+00:00"},{"alias_kind":"pith_short_8","alias_value":"UZ5I466Q","created_at":"2026-07-05T04:36:48.431977+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.19810","citing_title":"SimulS2ST-Omni: Data-Efficient Streaming Speech-to-Speech Translation via Explicit Trajectory Supervision","ref_index":205,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UZ5I466QO7ZFSEVU4HWHGP5CAP","json":"https://pith.science/pith/UZ5I466QO7ZFSEVU4HWHGP5CAP.json","graph_json":"https://pith.science/api/pith-number/UZ5I466QO7ZFSEVU4HWHGP5CAP/graph.json","events_json":"https://pith.science/api/pith-number/UZ5I466QO7ZFSEVU4HWHGP5CAP/events.json","paper":"https://pith.science/paper/UZ5I466Q"},"agent_actions":{"view_html":"https://pith.science/pith/UZ5I466QO7ZFSEVU4HWHGP5CAP","download_json":"https://pith.science/pith/UZ5I466QO7ZFSEVU4HWHGP5CAP.json","view_paper":"https://pith.science/paper/UZ5I466Q","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2204.05352&json=true","fetch_graph":"https://pith.science/api/pith-number/UZ5I466QO7ZFSEVU4HWHGP5CAP/graph.json","fetch_events":"https://pith.science/api/pith-number/UZ5I466QO7ZFSEVU4HWHGP5CAP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UZ5I466QO7ZFSEVU4HWHGP5CAP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UZ5I466QO7ZFSEVU4HWHGP5CAP/action/storage_attestation","attest_author":"https://pith.science/pith/UZ5I466QO7ZFSEVU4HWHGP5CAP/action/author_attestation","sign_citation":"https://pith.science/pith/UZ5I466QO7ZFSEVU4HWHGP5CAP/action/citation_signature","submit_replication":"https://pith.science/pith/UZ5I466QO7ZFSEVU4HWHGP5CAP/action/replication_record"}},"created_at":"2026-07-05T04:36:48.431977+00:00","updated_at":"2026-07-05T04:36:48.431977+00:00"}