{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:MMTRES5XBUNQUSJL342MJY4L5F","short_pith_number":"pith:MMTRES5X","schema_version":"1.0","canonical_sha256":"6327124bb70d1b0a492bdf34c4e38be978e3731e3c759fef6287ca0aae7d7b55","source":{"kind":"arxiv","id":"2505.22251","version":2},"attestation_state":"computed","paper":{"title":"Evaluation of LLMs in Speech is Often Flawed: Test Set Contamination in Large Language Models for Speech Recognition","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"eess.AS","authors_text":"Rogier Van Dalen, Shucong Zhang, Sourav Bhattacharya, Titouan Parcollet, Yuan Tseng","submitted_at":"2025-05-28T11:39:59Z","abstract_excerpt":"Recent work suggests that large language models (LLMs) can improve performance of speech tasks compared to existing systems. To support their claims, results on LibriSpeech and Common Voice are often quoted. However, this work finds that a substantial amount of the LibriSpeech and Common Voice evaluation sets appear in public LLM pretraining corpora. This calls into question the reliability of findings drawn from these two datasets. To measure contamination impact, LLMs trained with/without contamination are compared. A contaminated LLM is more likely to generate test sentences it has seen dur"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.22251","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"eess.AS","submitted_at":"2025-05-28T11:39:59Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"ba330f0584125e81a3d2173be2395ea6094c0d407c408503dc6f3bf87b110bef","abstract_canon_sha256":"a04e232a6fad094034b967af1d2ff7bb8c70f1d0b4dc3c7099294b3d818e1c9d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:16:35.549073Z","signature_b64":"1TdSRGtZGI79ZfqcZeW8i2cs2LCDh4WJyF/jh9Cz/saZmo+UAdfcKSYC0XETWjFVRjbtYAeD49HSD7PqESXOCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6327124bb70d1b0a492bdf34c4e38be978e3731e3c759fef6287ca0aae7d7b55","last_reissued_at":"2026-07-05T11:16:35.548376Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:16:35.548376Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluation of LLMs in Speech is Often Flawed: Test Set Contamination in Large Language Models for Speech Recognition","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"eess.AS","authors_text":"Rogier Van Dalen, Shucong Zhang, Sourav Bhattacharya, Titouan Parcollet, Yuan Tseng","submitted_at":"2025-05-28T11:39:59Z","abstract_excerpt":"Recent work suggests that large language models (LLMs) can improve performance of speech tasks compared to existing systems. To support their claims, results on LibriSpeech and Common Voice are often quoted. However, this work finds that a substantial amount of the LibriSpeech and Common Voice evaluation sets appear in public LLM pretraining corpora. This calls into question the reliability of findings drawn from these two datasets. To measure contamination impact, LLMs trained with/without contamination are compared. A contaminated LLM is more likely to generate test sentences it has seen dur"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.22251","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.22251/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.22251","created_at":"2026-07-05T11:16:35.548465+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.22251v2","created_at":"2026-07-05T11:16:35.548465+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.22251","created_at":"2026-07-05T11:16:35.548465+00:00"},{"alias_kind":"pith_short_12","alias_value":"MMTRES5XBUNQ","created_at":"2026-07-05T11:16:35.548465+00:00"},{"alias_kind":"pith_short_16","alias_value":"MMTRES5XBUNQUSJL","created_at":"2026-07-05T11:16:35.548465+00:00"},{"alias_kind":"pith_short_8","alias_value":"MMTRES5X","created_at":"2026-07-05T11:16:35.548465+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07608","citing_title":"Subtitle-Aligned Fine-Tuning of Whisper for Swiss German ASR: Benchmark Contamination, Convention Mismatch, and an Honest Baseline at 25.6% WER (13.8% cWER)","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2601.12248","citing_title":"AQUA-Bench: Beyond Finding Answers to Knowing When There Are None in Audio Question Answering","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25591","citing_title":"Walking Through Uncertainty: An Empirical Study of Uncertainty Estimation for Audio-Aware Large Language Models","ref_index":41,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MMTRES5XBUNQUSJL342MJY4L5F","json":"https://pith.science/pith/MMTRES5XBUNQUSJL342MJY4L5F.json","graph_json":"https://pith.science/api/pith-number/MMTRES5XBUNQUSJL342MJY4L5F/graph.json","events_json":"https://pith.science/api/pith-number/MMTRES5XBUNQUSJL342MJY4L5F/events.json","paper":"https://pith.science/paper/MMTRES5X"},"agent_actions":{"view_html":"https://pith.science/pith/MMTRES5XBUNQUSJL342MJY4L5F","download_json":"https://pith.science/pith/MMTRES5XBUNQUSJL342MJY4L5F.json","view_paper":"https://pith.science/paper/MMTRES5X","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.22251&json=true","fetch_graph":"https://pith.science/api/pith-number/MMTRES5XBUNQUSJL342MJY4L5F/graph.json","fetch_events":"https://pith.science/api/pith-number/MMTRES5XBUNQUSJL342MJY4L5F/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MMTRES5XBUNQUSJL342MJY4L5F/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MMTRES5XBUNQUSJL342MJY4L5F/action/storage_attestation","attest_author":"https://pith.science/pith/MMTRES5XBUNQUSJL342MJY4L5F/action/author_attestation","sign_citation":"https://pith.science/pith/MMTRES5XBUNQUSJL342MJY4L5F/action/citation_signature","submit_replication":"https://pith.science/pith/MMTRES5XBUNQUSJL342MJY4L5F/action/replication_record"}},"created_at":"2026-07-05T11:16:35.548465+00:00","updated_at":"2026-07-05T11:16:35.548465+00:00"}