{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:UPQEQKCK3FOBQ47W5OPZYTHQIS","short_pith_number":"pith:UPQEQKCK","schema_version":"1.0","canonical_sha256":"a3e048284ad95c1873f6eb9f9c4cf04480acbe6f285e1e79c1253fee9d9dfea8","source":{"kind":"arxiv","id":"2606.21237","version":1},"attestation_state":"computed","paper":{"title":"OpenWER: Improving Cross-Lingual ASR Evaluation and Enabling Token-Based Accuracy Metrics","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SD"],"primary_cat":"cs.CL","authors_text":"Gottfried Zimmermann, Korbinian Kuhn","submitted_at":"2026-06-19T09:00:48Z","abstract_excerpt":"Advances in deep learning and end-to-end Automatic Speech Recognition (ASR) have enabled robust multilingual models, but evaluation metrics remain limited in assessing accuracy. Efforts to improve or replace the common metric Word Error Rate (WER) often focus on English, leaving evaluations for low-resource languages under-explored and hindering fair cross-lingual comparisons. We present OpenWER, an open-source implementation that improves WER robustness through language-specific normalisation and compound word detection. A token-based Levenshtein alignment preserves complementary metrics and "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2606.21237","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2026-06-19T09:00:48Z","cross_cats_sorted":["cs.SD"],"title_canon_sha256":"fa3b7758cdb20e538878e3b1a110102277a58d8d9ebe35633cf2ea7a071c24b7","abstract_canon_sha256":"b5684dd80c9aa364a20030758a97e05da6fb85b53de37a71d289e1c4f01d25e5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-06-23T01:12:34.352494Z","signature_b64":"c7SrJZzHkf+Jvs7ATcN1r032QzlJm6asei2v0SyD0J/ML+m8Ecfdtu3lNMvIML2tYbNk0+4j+l3y+e6PvqBPBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a3e048284ad95c1873f6eb9f9c4cf04480acbe6f285e1e79c1253fee9d9dfea8","last_reissued_at":"2026-06-23T01:12:34.351979Z","signature_status":"signed_v1","first_computed_at":"2026-06-23T01:12:34.351979Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"OpenWER: Improving Cross-Lingual ASR Evaluation and Enabling Token-Based Accuracy Metrics","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SD"],"primary_cat":"cs.CL","authors_text":"Gottfried Zimmermann, Korbinian Kuhn","submitted_at":"2026-06-19T09:00:48Z","abstract_excerpt":"Advances in deep learning and end-to-end Automatic Speech Recognition (ASR) have enabled robust multilingual models, but evaluation metrics remain limited in assessing accuracy. Efforts to improve or replace the common metric Word Error Rate (WER) often focus on English, leaving evaluations for low-resource languages under-explored and hindering fair cross-lingual comparisons. We present OpenWER, an open-source implementation that improves WER robustness through language-specific normalisation and compound word detection. A token-based Levenshtein alignment preserves complementary metrics and "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2606.21237","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2606.21237/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2606.21237","created_at":"2026-06-23T01:12:34.352042+00:00"},{"alias_kind":"arxiv_version","alias_value":"2606.21237v1","created_at":"2026-06-23T01:12:34.352042+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2606.21237","created_at":"2026-06-23T01:12:34.352042+00:00"},{"alias_kind":"pith_short_12","alias_value":"UPQEQKCK3FOB","created_at":"2026-06-23T01:12:34.352042+00:00"},{"alias_kind":"pith_short_16","alias_value":"UPQEQKCK3FOBQ47W","created_at":"2026-06-23T01:12:34.352042+00:00"},{"alias_kind":"pith_short_8","alias_value":"UPQEQKCK","created_at":"2026-06-23T01:12:34.352042+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2606.21237","citing_title":"OpenWER: Improving Cross-Lingual ASR Evaluation and Enabling Token-Based Accuracy Metrics","ref_index":1,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UPQEQKCK3FOBQ47W5OPZYTHQIS","json":"https://pith.science/pith/UPQEQKCK3FOBQ47W5OPZYTHQIS.json","graph_json":"https://pith.science/api/pith-number/UPQEQKCK3FOBQ47W5OPZYTHQIS/graph.json","events_json":"https://pith.science/api/pith-number/UPQEQKCK3FOBQ47W5OPZYTHQIS/events.json","paper":"https://pith.science/paper/UPQEQKCK"},"agent_actions":{"view_html":"https://pith.science/pith/UPQEQKCK3FOBQ47W5OPZYTHQIS","download_json":"https://pith.science/pith/UPQEQKCK3FOBQ47W5OPZYTHQIS.json","view_paper":"https://pith.science/paper/UPQEQKCK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2606.21237&json=true","fetch_graph":"https://pith.science/api/pith-number/UPQEQKCK3FOBQ47W5OPZYTHQIS/graph.json","fetch_events":"https://pith.science/api/pith-number/UPQEQKCK3FOBQ47W5OPZYTHQIS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UPQEQKCK3FOBQ47W5OPZYTHQIS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UPQEQKCK3FOBQ47W5OPZYTHQIS/action/storage_attestation","attest_author":"https://pith.science/pith/UPQEQKCK3FOBQ47W5OPZYTHQIS/action/author_attestation","sign_citation":"https://pith.science/pith/UPQEQKCK3FOBQ47W5OPZYTHQIS/action/citation_signature","submit_replication":"https://pith.science/pith/UPQEQKCK3FOBQ47W5OPZYTHQIS/action/replication_record"}},"created_at":"2026-06-23T01:12:34.352042+00:00","updated_at":"2026-06-23T01:12:34.352042+00:00"}