{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XXPRU2SO5RX4GPK2AVWWYQH5EM","short_pith_number":"pith:XXPRU2SO","schema_version":"1.0","canonical_sha256":"bddf1a6a4eec6fc33d5a056d6c40fd232d59204390bdf9261037b4c0f7f7e970","source":{"kind":"arxiv","id":"2401.01572","version":1},"attestation_state":"computed","paper":{"title":"Hallucinations in Neural Automatic Speech Recognition: Identifying Errors and Hallucinatory Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Bertram E. Shi, Rita Frieske","submitted_at":"2024-01-03T06:56:56Z","abstract_excerpt":"Hallucinations are a type of output error produced by deep neural networks. While this has been studied in natural language processing, they have not been researched previously in automatic speech recognition. Here, we define hallucinations in ASR as transcriptions generated by a model that are semantically unrelated to the source utterance, yet still fluent and coherent. The similarity of hallucinations to probable natural language outputs of the model creates a danger of deception and impacts the credibility of the system. We show that commonly used metrics, such as word error rates, cannot "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.01572","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-01-03T06:56:56Z","cross_cats_sorted":["cs.SD","eess.AS"],"title_canon_sha256":"47306404207383aa5c63b989b43aedf3af98297cd1874d61f39ab1e655c524e2","abstract_canon_sha256":"2e220542f1d79ada390b7b72c2ed1f65c61254a7b06e2366f5f5329d9ac85b75"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:29:45.120851Z","signature_b64":"TxtSeQMpi9TP988JI4/ktOdXy2hmMkilLOgCOggEu3FdyK8O9hooaEoEAwR+dTZ9BXXdf+qD3cY1vqVtOOolCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bddf1a6a4eec6fc33d5a056d6c40fd232d59204390bdf9261037b4c0f7f7e970","last_reissued_at":"2026-07-05T07:29:45.120339Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:29:45.120339Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Hallucinations in Neural Automatic Speech Recognition: Identifying Errors and Hallucinatory Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Bertram E. Shi, Rita Frieske","submitted_at":"2024-01-03T06:56:56Z","abstract_excerpt":"Hallucinations are a type of output error produced by deep neural networks. While this has been studied in natural language processing, they have not been researched previously in automatic speech recognition. Here, we define hallucinations in ASR as transcriptions generated by a model that are semantically unrelated to the source utterance, yet still fluent and coherent. The similarity of hallucinations to probable natural language outputs of the model creates a danger of deception and impacts the credibility of the system. We show that commonly used metrics, such as word error rates, cannot "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.01572","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.01572/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.01572","created_at":"2026-07-05T07:29:45.120407+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.01572v1","created_at":"2026-07-05T07:29:45.120407+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.01572","created_at":"2026-07-05T07:29:45.120407+00:00"},{"alias_kind":"pith_short_12","alias_value":"XXPRU2SO5RX4","created_at":"2026-07-05T07:29:45.120407+00:00"},{"alias_kind":"pith_short_16","alias_value":"XXPRU2SO5RX4GPK2","created_at":"2026-07-05T07:29:45.120407+00:00"},{"alias_kind":"pith_short_8","alias_value":"XXPRU2SO","created_at":"2026-07-05T07:29:45.120407+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23060","citing_title":"From Text Metrics to Model Internals: A Study of Whisper ASR Hallucination Detection","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23048","citing_title":"HALAS: A Human-Annotated Dataset of Hallucinations of Modern ASR Systems","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2603.05094","citing_title":"TW-Sound580K: A Regional Audio-Text Dataset with Verification-Guided Curation for Localized Audio-Language Modeling","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2603.15045","citing_title":"LLMs and Speech: Integration vs. Combination","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12107","citing_title":"Too Good to Be True: A Study on Modern Automatic Speech Recognition for the Evaluation of Speech Enhancement","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21276","citing_title":"Do LLM Decoders Listen Fairly? Benchmarking How Language Model Priors Shape Bias in Speech Recognition","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19300","citing_title":"HalluAudio: A Comprehensive Benchmark for Hallucination Detection in Large Audio-Language Models","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19565","citing_title":"Detecting Hallucinations in SpeechLLMs at Inference Time Using Attention Maps","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XXPRU2SO5RX4GPK2AVWWYQH5EM","json":"https://pith.science/pith/XXPRU2SO5RX4GPK2AVWWYQH5EM.json","graph_json":"https://pith.science/api/pith-number/XXPRU2SO5RX4GPK2AVWWYQH5EM/graph.json","events_json":"https://pith.science/api/pith-number/XXPRU2SO5RX4GPK2AVWWYQH5EM/events.json","paper":"https://pith.science/paper/XXPRU2SO"},"agent_actions":{"view_html":"https://pith.science/pith/XXPRU2SO5RX4GPK2AVWWYQH5EM","download_json":"https://pith.science/pith/XXPRU2SO5RX4GPK2AVWWYQH5EM.json","view_paper":"https://pith.science/paper/XXPRU2SO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.01572&json=true","fetch_graph":"https://pith.science/api/pith-number/XXPRU2SO5RX4GPK2AVWWYQH5EM/graph.json","fetch_events":"https://pith.science/api/pith-number/XXPRU2SO5RX4GPK2AVWWYQH5EM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XXPRU2SO5RX4GPK2AVWWYQH5EM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XXPRU2SO5RX4GPK2AVWWYQH5EM/action/storage_attestation","attest_author":"https://pith.science/pith/XXPRU2SO5RX4GPK2AVWWYQH5EM/action/author_attestation","sign_citation":"https://pith.science/pith/XXPRU2SO5RX4GPK2AVWWYQH5EM/action/citation_signature","submit_replication":"https://pith.science/pith/XXPRU2SO5RX4GPK2AVWWYQH5EM/action/replication_record"}},"created_at":"2026-07-05T07:29:45.120407+00:00","updated_at":"2026-07-05T07:29:45.120407+00:00"}