{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2018:IBCEZFKZ3H47BO4T672ZATTLPW","short_pith_number":"pith:IBCEZFKZ","schema_version":"1.0","canonical_sha256":"40444c9559d9f9f0bb93f7f5904e6b7db24fa48d7c7e69a56c2f6923dbd939a1","source":{"kind":"arxiv","id":"1803.01090","version":1},"attestation_state":"computed","paper":{"title":"On Modular Training of Neural Acoustics-to-Word Model for LVCSR","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hao Li, Kai Yu, Qi Liu, Zhehuai Chen","submitted_at":"2018-03-03T02:08:46Z","abstract_excerpt":"End-to-end (E2E) automatic speech recognition (ASR) systems directly map acoustics to words using a unified model. Previous works mostly focus on E2E training a single model which integrates acoustic and language model into a whole. Although E2E training benefits from sequence modeling and simplified decoding pipelines, large amount of transcribed acoustic data is usually required, and traditional acoustic and language modelling techniques cannot be utilized. In this paper, a novel modular training framework of E2E ASR is proposed to separately train neural acoustic and language models during "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1803.01090","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2018-03-03T02:08:46Z","cross_cats_sorted":[],"title_canon_sha256":"288f1fd632629b321a63dbc4aaf14d586935ee4fb1a65b6410cefbe6eaa3b404","abstract_canon_sha256":"e39d75b3a25b41aa1fb3ec925f598e8e344ba6e7fe0dc62cff3fc7de55051959"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-18T00:22:01.791686Z","signature_b64":"6Q4e7U03N+BuVZpPiFKJgyHPrebsH7UG5ATHxnpGzvVltsCFjqR4q+TS9ndU0RY97eLeXj9PWyImLkWkeK6jCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"40444c9559d9f9f0bb93f7f5904e6b7db24fa48d7c7e69a56c2f6923dbd939a1","last_reissued_at":"2026-05-18T00:22:01.791130Z","signature_status":"signed_v1","first_computed_at":"2026-05-18T00:22:01.791130Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On Modular Training of Neural Acoustics-to-Word Model for LVCSR","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hao Li, Kai Yu, Qi Liu, Zhehuai Chen","submitted_at":"2018-03-03T02:08:46Z","abstract_excerpt":"End-to-end (E2E) automatic speech recognition (ASR) systems directly map acoustics to words using a unified model. Previous works mostly focus on E2E training a single model which integrates acoustic and language model into a whole. Although E2E training benefits from sequence modeling and simplified decoding pipelines, large amount of transcribed acoustic data is usually required, and traditional acoustic and language modelling techniques cannot be utilized. In this paper, a novel modular training framework of E2E ASR is proposed to separately train neural acoustic and language models during "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1803.01090","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1803.01090","created_at":"2026-05-18T00:22:01.791220+00:00"},{"alias_kind":"arxiv_version","alias_value":"1803.01090v1","created_at":"2026-05-18T00:22:01.791220+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1803.01090","created_at":"2026-05-18T00:22:01.791220+00:00"},{"alias_kind":"pith_short_12","alias_value":"IBCEZFKZ3H47","created_at":"2026-05-18T12:32:28.185984+00:00"},{"alias_kind":"pith_short_16","alias_value":"IBCEZFKZ3H47BO4T","created_at":"2026-05-18T12:32:28.185984+00:00"},{"alias_kind":"pith_short_8","alias_value":"IBCEZFKZ","created_at":"2026-05-18T12:32:28.185984+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IBCEZFKZ3H47BO4T672ZATTLPW","json":"https://pith.science/pith/IBCEZFKZ3H47BO4T672ZATTLPW.json","graph_json":"https://pith.science/api/pith-number/IBCEZFKZ3H47BO4T672ZATTLPW/graph.json","events_json":"https://pith.science/api/pith-number/IBCEZFKZ3H47BO4T672ZATTLPW/events.json","paper":"https://pith.science/paper/IBCEZFKZ"},"agent_actions":{"view_html":"https://pith.science/pith/IBCEZFKZ3H47BO4T672ZATTLPW","download_json":"https://pith.science/pith/IBCEZFKZ3H47BO4T672ZATTLPW.json","view_paper":"https://pith.science/paper/IBCEZFKZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1803.01090&json=true","fetch_graph":"https://pith.science/api/pith-number/IBCEZFKZ3H47BO4T672ZATTLPW/graph.json","fetch_events":"https://pith.science/api/pith-number/IBCEZFKZ3H47BO4T672ZATTLPW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IBCEZFKZ3H47BO4T672ZATTLPW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IBCEZFKZ3H47BO4T672ZATTLPW/action/storage_attestation","attest_author":"https://pith.science/pith/IBCEZFKZ3H47BO4T672ZATTLPW/action/author_attestation","sign_citation":"https://pith.science/pith/IBCEZFKZ3H47BO4T672ZATTLPW/action/citation_signature","submit_replication":"https://pith.science/pith/IBCEZFKZ3H47BO4T672ZATTLPW/action/replication_record"}},"created_at":"2026-05-18T00:22:01.791220+00:00","updated_at":"2026-05-18T00:22:01.791220+00:00"}