{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:N7Y5CIIT4QY2AEEMBVWEY62Z6S","short_pith_number":"pith:N7Y5CIIT","schema_version":"1.0","canonical_sha256":"6ff1d12113e431a0108c0d6c4c7b59f4811fb6669893a10ff3026fa3c792b20c","source":{"kind":"arxiv","id":"2111.09344","version":1},"attestation_state":"computed","paper":{"title":"The People's Speech: A Large-Scale Diverse English Speech Recognition Dataset for Commercial Usage","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Anjali Gopi, Daniel Galvez, David Kanter, Greg Diamos, Juan Ciro, Juan Felipe Cer\\'on, Keith Achorn, Mark Mazumder, Maximilian Lam, Vijay Janapa Reddi","submitted_at":"2021-11-17T19:14:40Z","abstract_excerpt":"The People's Speech is a free-to-download 30,000-hour and growing supervised conversational English speech recognition dataset licensed for academic and commercial usage under CC-BY-SA (with a CC-BY subset). The data is collected via searching the Internet for appropriately licensed audio data with existing transcriptions. We describe our data collection methodology and release our data collection system under the Apache 2.0 license. We show that a model trained on this dataset achieves a 9.98% word error rate on Librispeech's test-clean test set.Finally, we discuss the legal and ethical issue"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2111.09344","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-11-17T19:14:40Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"3b4d53a6952a2042262c9bdbe3d6b16b773e04fe77a1dc71fe422e28d01b9d61","abstract_canon_sha256":"e588971df9ba31e4e2852ec0902fa5afef54cea6f8eea4ee195b73160a78d50f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:33:10.274842Z","signature_b64":"LGJuJwl7iX8+nyVWdiDLcCgvrq3unppc8krOk++WwJfMFud0oJtqTXUeq4C++u5b2CF/UU4XD+cm+qP9bYZYCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6ff1d12113e431a0108c0d6c4c7b59f4811fb6669893a10ff3026fa3c792b20c","last_reissued_at":"2026-07-05T03:33:10.274346Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:33:10.274346Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The People's Speech: A Large-Scale Diverse English Speech Recognition Dataset for Commercial Usage","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Anjali Gopi, Daniel Galvez, David Kanter, Greg Diamos, Juan Ciro, Juan Felipe Cer\\'on, Keith Achorn, Mark Mazumder, Maximilian Lam, Vijay Janapa Reddi","submitted_at":"2021-11-17T19:14:40Z","abstract_excerpt":"The People's Speech is a free-to-download 30,000-hour and growing supervised conversational English speech recognition dataset licensed for academic and commercial usage under CC-BY-SA (with a CC-BY subset). The data is collected via searching the Internet for appropriately licensed audio data with existing transcriptions. We describe our data collection methodology and release our data collection system under the Apache 2.0 license. We show that a model trained on this dataset achieves a 9.98% word error rate on Librispeech's test-clean test set.Finally, we discuss the legal and ethical issue"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2111.09344","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2111.09344/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2111.09344","created_at":"2026-07-05T03:33:10.274407+00:00"},{"alias_kind":"arxiv_version","alias_value":"2111.09344v1","created_at":"2026-07-05T03:33:10.274407+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2111.09344","created_at":"2026-07-05T03:33:10.274407+00:00"},{"alias_kind":"pith_short_12","alias_value":"N7Y5CIIT4QY2","created_at":"2026-07-05T03:33:10.274407+00:00"},{"alias_kind":"pith_short_16","alias_value":"N7Y5CIIT4QY2AEEM","created_at":"2026-07-05T03:33:10.274407+00:00"},{"alias_kind":"pith_short_8","alias_value":"N7Y5CIIT","created_at":"2026-07-05T03:33:10.274407+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25391","citing_title":"From Sounds to Scenes: A Benchmark for Evaluating Context-Aware Auditory Scene Understanding in Large Audio Language Models","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22473","citing_title":"Interleaved Speech Language Models Latently Work In Text","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06837","citing_title":"SEAM: Shortcut-Aware Real-Time Detection of Scripted vs. Spontaneous Speech for Interview Guardrails","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09568","citing_title":"RADAR Challenge 2026: Robust Audio Deepfake Recognition under Media Transformations","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20830","citing_title":"Raon-OpenTTS: Open Models and Data for Robust Text-to-Speech","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09568","citing_title":"RADAR Challenge 2026: Robust Audio Deepfake Recognition under Media Transformations","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2509.22220","citing_title":"StableToken: A Noise-Robust Semantic Speech Tokenizer for Resilient SpeechLLMs","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12387","citing_title":"A Semi-Supervised Framework for Speech Confidence Detection using Whisper","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09568","citing_title":"RADAR Challenge 2026: Robust Audio Deepfake Recognition under Media Transformations","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08003","citing_title":"Rethinking Entropy Allocation in LLM-based ASR: Understanding the Dynamics between Speech Encoders and LLMs","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06765","citing_title":"VITA-QinYu: Expressive Spoken Language Model for Role-Playing and Singing","ref_index":125,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/N7Y5CIIT4QY2AEEMBVWEY62Z6S","json":"https://pith.science/pith/N7Y5CIIT4QY2AEEMBVWEY62Z6S.json","graph_json":"https://pith.science/api/pith-number/N7Y5CIIT4QY2AEEMBVWEY62Z6S/graph.json","events_json":"https://pith.science/api/pith-number/N7Y5CIIT4QY2AEEMBVWEY62Z6S/events.json","paper":"https://pith.science/paper/N7Y5CIIT"},"agent_actions":{"view_html":"https://pith.science/pith/N7Y5CIIT4QY2AEEMBVWEY62Z6S","download_json":"https://pith.science/pith/N7Y5CIIT4QY2AEEMBVWEY62Z6S.json","view_paper":"https://pith.science/paper/N7Y5CIIT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2111.09344&json=true","fetch_graph":"https://pith.science/api/pith-number/N7Y5CIIT4QY2AEEMBVWEY62Z6S/graph.json","fetch_events":"https://pith.science/api/pith-number/N7Y5CIIT4QY2AEEMBVWEY62Z6S/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/N7Y5CIIT4QY2AEEMBVWEY62Z6S/action/timestamp_anchor","attest_storage":"https://pith.science/pith/N7Y5CIIT4QY2AEEMBVWEY62Z6S/action/storage_attestation","attest_author":"https://pith.science/pith/N7Y5CIIT4QY2AEEMBVWEY62Z6S/action/author_attestation","sign_citation":"https://pith.science/pith/N7Y5CIIT4QY2AEEMBVWEY62Z6S/action/citation_signature","submit_replication":"https://pith.science/pith/N7Y5CIIT4QY2AEEMBVWEY62Z6S/action/replication_record"}},"created_at":"2026-07-05T03:33:10.274407+00:00","updated_at":"2026-07-05T03:33:10.274407+00:00"}