{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:JHWBA6GRYSLO3EA7VQXFHLR4TO","short_pith_number":"pith:JHWBA6GR","schema_version":"1.0","canonical_sha256":"49ec1078d1c496ed901fac2e53ae3c9b99313c1960a026e77590cb4a276f8329","source":{"kind":"arxiv","id":"2505.17070","version":1},"attestation_state":"computed","paper":{"title":"Improving endpoint detection in end-to-end streaming ASR for conversational speech","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Anandh C, Andreas Stolcke, Aravind Ganapathiraju, Jeena Prakash, Kadri Hacioglu, Karthik Pandia Durai, Manickavela Arumugam, Shankar Venkatesan, S.Pavankumar Dubagunta","submitted_at":"2025-05-19T15:19:59Z","abstract_excerpt":"ASR endpointing (EP) plays a major role in delivering a good user experience in products supporting human or artificial agents in human-human/machine conversations. Transducer-based ASR (T-ASR) is an end-to-end (E2E) ASR modelling technique preferred for streaming. A major limitation of T-ASR is delayed emission of ASR outputs, which could lead to errors or delays in EP. Inaccurate EP will cut the user off while speaking, returning incomplete transcript while delays in EP will increase the perceived latency, degrading the user experience. We propose methods to improve EP by addressing delayed "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.17070","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-19T15:19:59Z","cross_cats_sorted":["cs.AI","cs.SD","eess.AS"],"title_canon_sha256":"9f8cca2148d67e8b01e389eab472a297baf15648c8175cec810ed9363a2cf2e1","abstract_canon_sha256":"886e88673b1d27bd4f843e241b9bde367110b840eaffb3c3138e6db8a93b48f5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:07:46.580147Z","signature_b64":"RFVwbEmP1VP37wneY+9CRodNwV+b3xZi+ggWh2+5jOOg13aZ8ftTvVcDJP6SkrF03EdG3Q3WSZi3YlVKVFgoCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"49ec1078d1c496ed901fac2e53ae3c9b99313c1960a026e77590cb4a276f8329","last_reissued_at":"2026-07-05T11:07:46.579665Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:07:46.579665Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Improving endpoint detection in end-to-end streaming ASR for conversational speech","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Anandh C, Andreas Stolcke, Aravind Ganapathiraju, Jeena Prakash, Kadri Hacioglu, Karthik Pandia Durai, Manickavela Arumugam, Shankar Venkatesan, S.Pavankumar Dubagunta","submitted_at":"2025-05-19T15:19:59Z","abstract_excerpt":"ASR endpointing (EP) plays a major role in delivering a good user experience in products supporting human or artificial agents in human-human/machine conversations. Transducer-based ASR (T-ASR) is an end-to-end (E2E) ASR modelling technique preferred for streaming. A major limitation of T-ASR is delayed emission of ASR outputs, which could lead to errors or delays in EP. Inaccurate EP will cut the user off while speaking, returning incomplete transcript while delays in EP will increase the perceived latency, degrading the user experience. We propose methods to improve EP by addressing delayed "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.17070","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.17070/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.17070","created_at":"2026-07-05T11:07:46.579726+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.17070v1","created_at":"2026-07-05T11:07:46.579726+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.17070","created_at":"2026-07-05T11:07:46.579726+00:00"},{"alias_kind":"pith_short_12","alias_value":"JHWBA6GRYSLO","created_at":"2026-07-05T11:07:46.579726+00:00"},{"alias_kind":"pith_short_16","alias_value":"JHWBA6GRYSLO3EA7","created_at":"2026-07-05T11:07:46.579726+00:00"},{"alias_kind":"pith_short_8","alias_value":"JHWBA6GR","created_at":"2026-07-05T11:07:46.579726+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.17070","citing_title":"Improving endpoint detection in end-to-end streaming ASR for conversational speech","ref_index":1,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JHWBA6GRYSLO3EA7VQXFHLR4TO","json":"https://pith.science/pith/JHWBA6GRYSLO3EA7VQXFHLR4TO.json","graph_json":"https://pith.science/api/pith-number/JHWBA6GRYSLO3EA7VQXFHLR4TO/graph.json","events_json":"https://pith.science/api/pith-number/JHWBA6GRYSLO3EA7VQXFHLR4TO/events.json","paper":"https://pith.science/paper/JHWBA6GR"},"agent_actions":{"view_html":"https://pith.science/pith/JHWBA6GRYSLO3EA7VQXFHLR4TO","download_json":"https://pith.science/pith/JHWBA6GRYSLO3EA7VQXFHLR4TO.json","view_paper":"https://pith.science/paper/JHWBA6GR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.17070&json=true","fetch_graph":"https://pith.science/api/pith-number/JHWBA6GRYSLO3EA7VQXFHLR4TO/graph.json","fetch_events":"https://pith.science/api/pith-number/JHWBA6GRYSLO3EA7VQXFHLR4TO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JHWBA6GRYSLO3EA7VQXFHLR4TO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JHWBA6GRYSLO3EA7VQXFHLR4TO/action/storage_attestation","attest_author":"https://pith.science/pith/JHWBA6GRYSLO3EA7VQXFHLR4TO/action/author_attestation","sign_citation":"https://pith.science/pith/JHWBA6GRYSLO3EA7VQXFHLR4TO/action/citation_signature","submit_replication":"https://pith.science/pith/JHWBA6GRYSLO3EA7VQXFHLR4TO/action/replication_record"}},"created_at":"2026-07-05T11:07:46.579726+00:00","updated_at":"2026-07-05T11:07:46.579726+00:00"}