{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:V4SKWMOI2433LR62CBCRNVZPHI","short_pith_number":"pith:V4SKWMOI","schema_version":"1.0","canonical_sha256":"af24ab31c8d737b5c7da104516d72f3a2c9294bb9232e016a7e83eb9036e8b0d","source":{"kind":"arxiv","id":"2005.09684","version":2},"attestation_state":"computed","paper":{"title":"Exploring Transformers for Large-Scale Speech Recognition","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.SD"],"primary_cat":"eess.AS","authors_text":"Changliang Liu, Jinyu Li, Liang Lu, Yifan Gong","submitted_at":"2020-05-19T18:07:14Z","abstract_excerpt":"While recurrent neural networks still largely define state-of-the-art speech recognition systems, the Transformer network has been proven to be a competitive alternative, especially in the offline condition. Most studies with Transformers have been constrained in a relatively small scale setting, and some forms of data argumentation approaches are usually applied to combat the data sparsity issue. In this paper, we aim at understanding the behaviors of Transformers in the large-scale speech recognition setting, where we have used around 65,000 hours of training data. We investigated various as"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2005.09684","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.AS","submitted_at":"2020-05-19T18:07:14Z","cross_cats_sorted":["cs.CL","cs.SD"],"title_canon_sha256":"c185edcffa8d4678fe85b0fde8b0174b9ebfb97d01afa5871c85d7379d0b95ce","abstract_canon_sha256":"e4ee2695fd2f03567d4d5f50d4a4387636867387806f7759182c166d9c588300"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:26:30.942755Z","signature_b64":"tgGrLbwH11i3sXEcP56Nyhb1Mi5nRLDyyJC5v/hcsGa434BVh2LF2pFgqvE6FHiYJXVJ8l21oMF9f7q21r2KCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"af24ab31c8d737b5c7da104516d72f3a2c9294bb9232e016a7e83eb9036e8b0d","last_reissued_at":"2026-07-05T01:26:30.942281Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:26:30.942281Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Exploring Transformers for Large-Scale Speech Recognition","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.SD"],"primary_cat":"eess.AS","authors_text":"Changliang Liu, Jinyu Li, Liang Lu, Yifan Gong","submitted_at":"2020-05-19T18:07:14Z","abstract_excerpt":"While recurrent neural networks still largely define state-of-the-art speech recognition systems, the Transformer network has been proven to be a competitive alternative, especially in the offline condition. Most studies with Transformers have been constrained in a relatively small scale setting, and some forms of data argumentation approaches are usually applied to combat the data sparsity issue. In this paper, we aim at understanding the behaviors of Transformers in the large-scale speech recognition setting, where we have used around 65,000 hours of training data. We investigated various as"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2005.09684","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2005.09684/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2005.09684","created_at":"2026-07-05T01:26:30.942337+00:00"},{"alias_kind":"arxiv_version","alias_value":"2005.09684v2","created_at":"2026-07-05T01:26:30.942337+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2005.09684","created_at":"2026-07-05T01:26:30.942337+00:00"},{"alias_kind":"pith_short_12","alias_value":"V4SKWMOI2433","created_at":"2026-07-05T01:26:30.942337+00:00"},{"alias_kind":"pith_short_16","alias_value":"V4SKWMOI2433LR62","created_at":"2026-07-05T01:26:30.942337+00:00"},{"alias_kind":"pith_short_8","alias_value":"V4SKWMOI","created_at":"2026-07-05T01:26:30.942337+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.09957","citing_title":"Which one Performs Better? Wav2Vec or Whisper? Applying both in Badini Kurdish Speech to Text (BKSTT)","ref_index":13,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/V4SKWMOI2433LR62CBCRNVZPHI","json":"https://pith.science/pith/V4SKWMOI2433LR62CBCRNVZPHI.json","graph_json":"https://pith.science/api/pith-number/V4SKWMOI2433LR62CBCRNVZPHI/graph.json","events_json":"https://pith.science/api/pith-number/V4SKWMOI2433LR62CBCRNVZPHI/events.json","paper":"https://pith.science/paper/V4SKWMOI"},"agent_actions":{"view_html":"https://pith.science/pith/V4SKWMOI2433LR62CBCRNVZPHI","download_json":"https://pith.science/pith/V4SKWMOI2433LR62CBCRNVZPHI.json","view_paper":"https://pith.science/paper/V4SKWMOI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2005.09684&json=true","fetch_graph":"https://pith.science/api/pith-number/V4SKWMOI2433LR62CBCRNVZPHI/graph.json","fetch_events":"https://pith.science/api/pith-number/V4SKWMOI2433LR62CBCRNVZPHI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/V4SKWMOI2433LR62CBCRNVZPHI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/V4SKWMOI2433LR62CBCRNVZPHI/action/storage_attestation","attest_author":"https://pith.science/pith/V4SKWMOI2433LR62CBCRNVZPHI/action/author_attestation","sign_citation":"https://pith.science/pith/V4SKWMOI2433LR62CBCRNVZPHI/action/citation_signature","submit_replication":"https://pith.science/pith/V4SKWMOI2433LR62CBCRNVZPHI/action/replication_record"}},"created_at":"2026-07-05T01:26:30.942337+00:00","updated_at":"2026-07-05T01:26:30.942337+00:00"}