{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:VWINZ6JDAR767LKTU5WXWO7X3W","short_pith_number":"pith:VWINZ6JD","schema_version":"1.0","canonical_sha256":"ad90dcf923047fefad53a76d7b3bf7ddbdddadf016b4ffdf2b501499932545d0","source":{"kind":"arxiv","id":"2509.09987","version":1},"attestation_state":"computed","paper":{"title":"Whisper Has an Internal Word Aligner","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"eess.AS","authors_text":"Hao Tang, Sung-Lin Yeh, Yen Meng","submitted_at":"2025-09-12T06:03:24Z","abstract_excerpt":"There is an increasing interest in obtaining accurate word-level timestamps from strong automatic speech recognizers, in particular Whisper. Existing approaches either require additional training or are simply not competitive. The evaluation in prior work is also relatively loose, typically using a tolerance of more than 200 ms. In this work, we discover attention heads in Whisper that capture accurate word alignments and are distinctively different from those that do not. Moreover, we find that using characters produces finer and more accurate alignments than using wordpieces. Based on these "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2509.09987","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.AS","submitted_at":"2025-09-12T06:03:24Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"c7a909cb57df89cebf74178b8914e6911ede1770ee4bc58a256e01512d74688a","abstract_canon_sha256":"0e4c7f08232a9c834105cf1bb90d10cd5a4ba8306d2b3d965298697a2a7690b4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:09:53.245943Z","signature_b64":"El1m5y4Fd6ELrxAtZey5bpE3NqFGVsbibeGaOF1A8fs8LHNcEKSQ02AFtrRuN3TSzNQokmGGKkaSXnHkT8TrDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ad90dcf923047fefad53a76d7b3bf7ddbdddadf016b4ffdf2b501499932545d0","last_reissued_at":"2026-07-05T12:09:53.245464Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:09:53.245464Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Whisper Has an Internal Word Aligner","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"eess.AS","authors_text":"Hao Tang, Sung-Lin Yeh, Yen Meng","submitted_at":"2025-09-12T06:03:24Z","abstract_excerpt":"There is an increasing interest in obtaining accurate word-level timestamps from strong automatic speech recognizers, in particular Whisper. Existing approaches either require additional training or are simply not competitive. The evaluation in prior work is also relatively loose, typically using a tolerance of more than 200 ms. In this work, we discover attention heads in Whisper that capture accurate word alignments and are distinctively different from those that do not. Moreover, we find that using characters produces finer and more accurate alignments than using wordpieces. Based on these "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.09987","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.09987/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2509.09987","created_at":"2026-07-05T12:09:53.245526+00:00"},{"alias_kind":"arxiv_version","alias_value":"2509.09987v1","created_at":"2026-07-05T12:09:53.245526+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.09987","created_at":"2026-07-05T12:09:53.245526+00:00"},{"alias_kind":"pith_short_12","alias_value":"VWINZ6JDAR76","created_at":"2026-07-05T12:09:53.245526+00:00"},{"alias_kind":"pith_short_16","alias_value":"VWINZ6JDAR767LKT","created_at":"2026-07-05T12:09:53.245526+00:00"},{"alias_kind":"pith_short_8","alias_value":"VWINZ6JD","created_at":"2026-07-05T12:09:53.245526+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05364","citing_title":"REDDIT: Correcting Model-Generated Timestamp Drift in ASR without Forgetting via Replay-Based Distribution Editing","ref_index":18,"is_internal_anchor":true},{"citing_arxiv_id":"2605.23604","citing_title":"Word-Level Modeling with Alignment-Aware Acoustic Fusion for Text-Assisted Intelligibility Prediction in Listeners with Hearing Loss","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05231","citing_title":"Prompting Whisper for Joint Speech Transcription and Diarization","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VWINZ6JDAR767LKTU5WXWO7X3W","json":"https://pith.science/pith/VWINZ6JDAR767LKTU5WXWO7X3W.json","graph_json":"https://pith.science/api/pith-number/VWINZ6JDAR767LKTU5WXWO7X3W/graph.json","events_json":"https://pith.science/api/pith-number/VWINZ6JDAR767LKTU5WXWO7X3W/events.json","paper":"https://pith.science/paper/VWINZ6JD"},"agent_actions":{"view_html":"https://pith.science/pith/VWINZ6JDAR767LKTU5WXWO7X3W","download_json":"https://pith.science/pith/VWINZ6JDAR767LKTU5WXWO7X3W.json","view_paper":"https://pith.science/paper/VWINZ6JD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2509.09987&json=true","fetch_graph":"https://pith.science/api/pith-number/VWINZ6JDAR767LKTU5WXWO7X3W/graph.json","fetch_events":"https://pith.science/api/pith-number/VWINZ6JDAR767LKTU5WXWO7X3W/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VWINZ6JDAR767LKTU5WXWO7X3W/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VWINZ6JDAR767LKTU5WXWO7X3W/action/storage_attestation","attest_author":"https://pith.science/pith/VWINZ6JDAR767LKTU5WXWO7X3W/action/author_attestation","sign_citation":"https://pith.science/pith/VWINZ6JDAR767LKTU5WXWO7X3W/action/citation_signature","submit_replication":"https://pith.science/pith/VWINZ6JDAR767LKTU5WXWO7X3W/action/replication_record"}},"created_at":"2026-07-05T12:09:53.245526+00:00","updated_at":"2026-07-05T12:09:53.245526+00:00"}