{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:MRE24FMQ5AXQC2EZ4IF4JLRKLA","short_pith_number":"pith:MRE24FMQ","schema_version":"1.0","canonical_sha256":"6449ae1590e82f016899e20bc4ae2a580c4c0e10cb8b94b47831e156e2107cf6","source":{"kind":"arxiv","id":"1904.08779","version":3},"attestation_state":"computed","paper":{"title":"SpecAugment: A Simple Data Augmentation Method for Automatic Speech Recognition","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG","cs.SD","stat.ML"],"primary_cat":"eess.AS","authors_text":"Barret Zoph, Chung-Cheng Chiu, Daniel S. Park, Ekin D. Cubuk, Quoc V. Le, William Chan, Yu Zhang","submitted_at":"2019-04-18T17:53:38Z","abstract_excerpt":"We present SpecAugment, a simple data augmentation method for speech recognition. SpecAugment is applied directly to the feature inputs of a neural network (i.e., filter bank coefficients). The augmentation policy consists of warping the features, masking blocks of frequency channels, and masking blocks of time steps. We apply SpecAugment on Listen, Attend and Spell networks for end-to-end speech recognition tasks. We achieve state-of-the-art performance on the LibriSpeech 960h and Swichboard 300h tasks, outperforming all prior work. On LibriSpeech, we achieve 6.8% WER on test-other without th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1904.08779","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.AS","submitted_at":"2019-04-18T17:53:38Z","cross_cats_sorted":["cs.CL","cs.LG","cs.SD","stat.ML"],"title_canon_sha256":"583ae475d632a2fa684daa33edf6623ff3d9db05f80cb3be05c5e7392ebe41a7","abstract_canon_sha256":"41e0643c597deda714981e713a954c99d3945788ca1a76e2f025ef43643bcf34"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:23:21.807211Z","signature_b64":"ugY7TaL2vBbWUucOESc/JX64sVwbq6VHVKMruRDmEBlfyUrXlGagov46WUU45uOtCy4qtm3JK2WjVHOjzHXqAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6449ae1590e82f016899e20bc4ae2a580c4c0e10cb8b94b47831e156e2107cf6","last_reissued_at":"2026-07-05T00:23:21.806706Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:23:21.806706Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SpecAugment: A Simple Data Augmentation Method for Automatic Speech Recognition","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG","cs.SD","stat.ML"],"primary_cat":"eess.AS","authors_text":"Barret Zoph, Chung-Cheng Chiu, Daniel S. Park, Ekin D. Cubuk, Quoc V. Le, William Chan, Yu Zhang","submitted_at":"2019-04-18T17:53:38Z","abstract_excerpt":"We present SpecAugment, a simple data augmentation method for speech recognition. SpecAugment is applied directly to the feature inputs of a neural network (i.e., filter bank coefficients). The augmentation policy consists of warping the features, masking blocks of frequency channels, and masking blocks of time steps. We apply SpecAugment on Listen, Attend and Spell networks for end-to-end speech recognition tasks. We achieve state-of-the-art performance on the LibriSpeech 960h and Swichboard 300h tasks, outperforming all prior work. On LibriSpeech, we achieve 6.8% WER on test-other without th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1904.08779","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1904.08779/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1904.08779","created_at":"2026-07-05T00:23:21.806783+00:00"},{"alias_kind":"arxiv_version","alias_value":"1904.08779v3","created_at":"2026-07-05T00:23:21.806783+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1904.08779","created_at":"2026-07-05T00:23:21.806783+00:00"},{"alias_kind":"pith_short_12","alias_value":"MRE24FMQ5AXQ","created_at":"2026-07-05T00:23:21.806783+00:00"},{"alias_kind":"pith_short_16","alias_value":"MRE24FMQ5AXQC2EZ","created_at":"2026-07-05T00:23:21.806783+00:00"},{"alias_kind":"pith_short_8","alias_value":"MRE24FMQ","created_at":"2026-07-05T00:23:21.806783+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":18,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2607.05165","citing_title":"Physiological Noise Augmentation Improves Non-Invasive Brain-to-Speech","ref_index":33,"is_internal_anchor":true},{"citing_arxiv_id":"2604.18105","citing_title":"NIM4-ASR: Towards Efficient, Robust, and Customizable Real-Time LLM-Based ASR","ref_index":16,"is_internal_anchor":true},{"citing_arxiv_id":"2606.18659","citing_title":"Responsible ASR: Overcoming Challenges of Foundational Models in Narrow-Band and Low-Resource Settings","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08486","citing_title":"TRADE: Transducer-Augmented Decoder for Speech LLM","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00387","citing_title":"From Objectives to Applications: Aligning Architectural Biases in Audio Self-Supervised Learning","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02212","citing_title":"C2GA: A Class-Controllable Generative Augmentation Framework for Respiratory Sound Classification","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23954","citing_title":"EchoDistill:Alignment Noisy-to-Clean Self-Distillation for Robust Audio LLMs","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29862","citing_title":"Mitigating Stethoscope-Induced Shortcuts in Respiratory Sound Classification under Federated Domain Generalization with Causality-Inspired Interventions","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2309.12802","citing_title":"Deepfake audio as a data augmentation technique for training automatic speech to text transcription models","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2411.12363","citing_title":"DGSNA: Dynamic Generative Scene-based Noise Addition method","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19632","citing_title":"Executable Boundary Contracts for Sound Event Traces","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16555","citing_title":"MedASR: An Open-Source Model for High-Accuracy Medical Dictation","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2602.12783","citing_title":"SQuTR: A Robustness Benchmark for Spoken Query to Text Retrieval under Acoustic Noise","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2603.29087","citing_title":"IQRA 2026: Interspeech Challenge on Automatic Pronunciation Assessment for Modern Standard Arabic (MSA)","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18105","citing_title":"NIM4-ASR: Towards Efficient, Robust, and Customizable Real-Time LLM-Based ASR","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08528","citing_title":"A-SLIP: Acoustic Sensing for Continuous In-hand Slip Estimation","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05830","citing_title":"\"OK Aura, Be Fair With Me\": Demographics-Agnostic Training for Bias Mitigation in Wake-up Word Detection","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24770","citing_title":"Elderly-Contextual Data Augmentation via Speech Synthesis for Elderly ASR","ref_index":40,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MRE24FMQ5AXQC2EZ4IF4JLRKLA","json":"https://pith.science/pith/MRE24FMQ5AXQC2EZ4IF4JLRKLA.json","graph_json":"https://pith.science/api/pith-number/MRE24FMQ5AXQC2EZ4IF4JLRKLA/graph.json","events_json":"https://pith.science/api/pith-number/MRE24FMQ5AXQC2EZ4IF4JLRKLA/events.json","paper":"https://pith.science/paper/MRE24FMQ"},"agent_actions":{"view_html":"https://pith.science/pith/MRE24FMQ5AXQC2EZ4IF4JLRKLA","download_json":"https://pith.science/pith/MRE24FMQ5AXQC2EZ4IF4JLRKLA.json","view_paper":"https://pith.science/paper/MRE24FMQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1904.08779&json=true","fetch_graph":"https://pith.science/api/pith-number/MRE24FMQ5AXQC2EZ4IF4JLRKLA/graph.json","fetch_events":"https://pith.science/api/pith-number/MRE24FMQ5AXQC2EZ4IF4JLRKLA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MRE24FMQ5AXQC2EZ4IF4JLRKLA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MRE24FMQ5AXQC2EZ4IF4JLRKLA/action/storage_attestation","attest_author":"https://pith.science/pith/MRE24FMQ5AXQC2EZ4IF4JLRKLA/action/author_attestation","sign_citation":"https://pith.science/pith/MRE24FMQ5AXQC2EZ4IF4JLRKLA/action/citation_signature","submit_replication":"https://pith.science/pith/MRE24FMQ5AXQC2EZ4IF4JLRKLA/action/replication_record"}},"created_at":"2026-07-05T00:23:21.806783+00:00","updated_at":"2026-07-05T00:23:21.806783+00:00"}