{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:PSLZ26BL4CR4EBIMR5ZXUCDCOQ","short_pith_number":"pith:PSLZ26BL","schema_version":"1.0","canonical_sha256":"7c979d782be0a3c2050c8f737a086274168a602ee3485a0267723f2288293745","source":{"kind":"arxiv","id":"1911.08460","version":3},"attestation_state":"computed","paper":{"title":"End-to-end ASR: from Supervised to Semi-Supervised Learning with Modern Architectures","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Anuroop Sriram, Edouard Grave, Gabriel Synnaeve, Jacob Kahn, Qiantong Xu, Ronan Collobert, Tatiana Likhomanenko, Vineel Pratap, Vitaliy Liptchinsky","submitted_at":"2019-11-19T18:40:02Z","abstract_excerpt":"We study pseudo-labeling for the semi-supervised training of ResNet, Time-Depth Separable ConvNets, and Transformers for speech recognition, with either CTC or Seq2Seq loss functions. We perform experiments on the standard LibriSpeech dataset, and leverage additional unlabeled data from LibriVox through pseudo-labeling. We show that while Transformer-based acoustic models have superior performance with the supervised dataset alone, semi-supervision improves all models across architectures and loss functions and bridges much of the performance gaps between them. In doing so, we reach a new stat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1911.08460","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2019-11-19T18:40:02Z","cross_cats_sorted":["cs.SD","eess.AS"],"title_canon_sha256":"4f3343b557d87e45615c834a4fafb3f86c6f375e0b088214363a2c9434eab262","abstract_canon_sha256":"a9ee36201c656354534fa4e9875c1f0fc588dbc98846d7ed068a7ba1c56b5968"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:19:25.392078Z","signature_b64":"346k6rAmjHuROk4Te+q9ZaD9HlkBoiIIt80q3RMmb0QMZPlB9CxjqDU9sqY8JCMaa37MpUmFj4vkITZ5Q3VxBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7c979d782be0a3c2050c8f737a086274168a602ee3485a0267723f2288293745","last_reissued_at":"2026-07-05T01:19:25.391641Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:19:25.391641Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"End-to-end ASR: from Supervised to Semi-Supervised Learning with Modern Architectures","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Anuroop Sriram, Edouard Grave, Gabriel Synnaeve, Jacob Kahn, Qiantong Xu, Ronan Collobert, Tatiana Likhomanenko, Vineel Pratap, Vitaliy Liptchinsky","submitted_at":"2019-11-19T18:40:02Z","abstract_excerpt":"We study pseudo-labeling for the semi-supervised training of ResNet, Time-Depth Separable ConvNets, and Transformers for speech recognition, with either CTC or Seq2Seq loss functions. We perform experiments on the standard LibriSpeech dataset, and leverage additional unlabeled data from LibriVox through pseudo-labeling. We show that while Transformer-based acoustic models have superior performance with the supervised dataset alone, semi-supervision improves all models across architectures and loss functions and bridges much of the performance gaps between them. In doing so, we reach a new stat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1911.08460","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1911.08460/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1911.08460","created_at":"2026-07-05T01:19:25.391707+00:00"},{"alias_kind":"arxiv_version","alias_value":"1911.08460v3","created_at":"2026-07-05T01:19:25.391707+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1911.08460","created_at":"2026-07-05T01:19:25.391707+00:00"},{"alias_kind":"pith_short_12","alias_value":"PSLZ26BL4CR4","created_at":"2026-07-05T01:19:25.391707+00:00"},{"alias_kind":"pith_short_16","alias_value":"PSLZ26BL4CR4EBIM","created_at":"2026-07-05T01:19:25.391707+00:00"},{"alias_kind":"pith_short_8","alias_value":"PSLZ26BL","created_at":"2026-07-05T01:19:25.391707+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.22810","citing_title":"A Self-Training Approach for Whisper to Enhance Long Dysarthric Speech Recognition","ref_index":36,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PSLZ26BL4CR4EBIMR5ZXUCDCOQ","json":"https://pith.science/pith/PSLZ26BL4CR4EBIMR5ZXUCDCOQ.json","graph_json":"https://pith.science/api/pith-number/PSLZ26BL4CR4EBIMR5ZXUCDCOQ/graph.json","events_json":"https://pith.science/api/pith-number/PSLZ26BL4CR4EBIMR5ZXUCDCOQ/events.json","paper":"https://pith.science/paper/PSLZ26BL"},"agent_actions":{"view_html":"https://pith.science/pith/PSLZ26BL4CR4EBIMR5ZXUCDCOQ","download_json":"https://pith.science/pith/PSLZ26BL4CR4EBIMR5ZXUCDCOQ.json","view_paper":"https://pith.science/paper/PSLZ26BL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1911.08460&json=true","fetch_graph":"https://pith.science/api/pith-number/PSLZ26BL4CR4EBIMR5ZXUCDCOQ/graph.json","fetch_events":"https://pith.science/api/pith-number/PSLZ26BL4CR4EBIMR5ZXUCDCOQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PSLZ26BL4CR4EBIMR5ZXUCDCOQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PSLZ26BL4CR4EBIMR5ZXUCDCOQ/action/storage_attestation","attest_author":"https://pith.science/pith/PSLZ26BL4CR4EBIMR5ZXUCDCOQ/action/author_attestation","sign_citation":"https://pith.science/pith/PSLZ26BL4CR4EBIMR5ZXUCDCOQ/action/citation_signature","submit_replication":"https://pith.science/pith/PSLZ26BL4CR4EBIMR5ZXUCDCOQ/action/replication_record"}},"created_at":"2026-07-05T01:19:25.391707+00:00","updated_at":"2026-07-05T01:19:25.391707+00:00"}