{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:G3YJSDYLXKPOM4YA5AOUUGGIWE","short_pith_number":"pith:G3YJSDYL","schema_version":"1.0","canonical_sha256":"36f0990f0bba9ee67300e81d4a18c8b11412501a8d29b20cf67b517852d916ab","source":{"kind":"arxiv","id":"2010.04284","version":1},"attestation_state":"computed","paper":{"title":"Leveraging Unpaired Text Data for Training End-to-End Speech-to-Intent Systems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Brian Kingsbury, Hong-Kwang Kuo, Kartik Audhkhasi, Michael Picheny, Ron Hoory, Samuel Thomas, Yinghui Huang, Zvi Kons","submitted_at":"2020-10-08T22:16:26Z","abstract_excerpt":"Training an end-to-end (E2E) neural network speech-to-intent (S2I) system that directly extracts intents from speech requires large amounts of intent-labeled speech data, which is time consuming and expensive to collect. Initializing the S2I model with an ASR model trained on copious speech data can alleviate data sparsity. In this paper, we attempt to leverage NLU text resources. We implemented a CTC-based S2I system that matches the performance of a state-of-the-art, traditional cascaded SLU system. We performed controlled experiments with varying amounts of speech and text training data. Wh"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2010.04284","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2020-10-08T22:16:26Z","cross_cats_sorted":["cs.LG","cs.SD","eess.AS"],"title_canon_sha256":"c451223e14a3a1942bc5387a8d71f55c91a9da08683fba4ac7d999c38111463e","abstract_canon_sha256":"0a8126d20d9ca1a917d95a71e5e18ee03456635f14d357a9669d9b3fcece68cf"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:41:35.276578Z","signature_b64":"fuS/hybOTUh57isymGqBTG/s4jz0F2Kdv/m+6vyQZqQzVhwKpAJTvbLUjVSuVmg4LrSXdOopVPrcBVDdSSxKCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"36f0990f0bba9ee67300e81d4a18c8b11412501a8d29b20cf67b517852d916ab","last_reissued_at":"2026-07-05T01:41:35.276133Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:41:35.276133Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Leveraging Unpaired Text Data for Training End-to-End Speech-to-Intent Systems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Brian Kingsbury, Hong-Kwang Kuo, Kartik Audhkhasi, Michael Picheny, Ron Hoory, Samuel Thomas, Yinghui Huang, Zvi Kons","submitted_at":"2020-10-08T22:16:26Z","abstract_excerpt":"Training an end-to-end (E2E) neural network speech-to-intent (S2I) system that directly extracts intents from speech requires large amounts of intent-labeled speech data, which is time consuming and expensive to collect. Initializing the S2I model with an ASR model trained on copious speech data can alleviate data sparsity. In this paper, we attempt to leverage NLU text resources. We implemented a CTC-based S2I system that matches the performance of a state-of-the-art, traditional cascaded SLU system. We performed controlled experiments with varying amounts of speech and text training data. Wh"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2010.04284","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2010.04284/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2010.04284","created_at":"2026-07-05T01:41:35.276200+00:00"},{"alias_kind":"arxiv_version","alias_value":"2010.04284v1","created_at":"2026-07-05T01:41:35.276200+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2010.04284","created_at":"2026-07-05T01:41:35.276200+00:00"},{"alias_kind":"pith_short_12","alias_value":"G3YJSDYLXKPO","created_at":"2026-07-05T01:41:35.276200+00:00"},{"alias_kind":"pith_short_16","alias_value":"G3YJSDYLXKPOM4YA","created_at":"2026-07-05T01:41:35.276200+00:00"},{"alias_kind":"pith_short_8","alias_value":"G3YJSDYL","created_at":"2026-07-05T01:41:35.276200+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/G3YJSDYLXKPOM4YA5AOUUGGIWE","json":"https://pith.science/pith/G3YJSDYLXKPOM4YA5AOUUGGIWE.json","graph_json":"https://pith.science/api/pith-number/G3YJSDYLXKPOM4YA5AOUUGGIWE/graph.json","events_json":"https://pith.science/api/pith-number/G3YJSDYLXKPOM4YA5AOUUGGIWE/events.json","paper":"https://pith.science/paper/G3YJSDYL"},"agent_actions":{"view_html":"https://pith.science/pith/G3YJSDYLXKPOM4YA5AOUUGGIWE","download_json":"https://pith.science/pith/G3YJSDYLXKPOM4YA5AOUUGGIWE.json","view_paper":"https://pith.science/paper/G3YJSDYL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2010.04284&json=true","fetch_graph":"https://pith.science/api/pith-number/G3YJSDYLXKPOM4YA5AOUUGGIWE/graph.json","fetch_events":"https://pith.science/api/pith-number/G3YJSDYLXKPOM4YA5AOUUGGIWE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/G3YJSDYLXKPOM4YA5AOUUGGIWE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/G3YJSDYLXKPOM4YA5AOUUGGIWE/action/storage_attestation","attest_author":"https://pith.science/pith/G3YJSDYLXKPOM4YA5AOUUGGIWE/action/author_attestation","sign_citation":"https://pith.science/pith/G3YJSDYLXKPOM4YA5AOUUGGIWE/action/citation_signature","submit_replication":"https://pith.science/pith/G3YJSDYLXKPOM4YA5AOUUGGIWE/action/replication_record"}},"created_at":"2026-07-05T01:41:35.276200+00:00","updated_at":"2026-07-05T01:41:35.276200+00:00"}