{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:Q3KA24K4TYACSISCWNI47J3P26","short_pith_number":"pith:Q3KA24K4","schema_version":"1.0","canonical_sha256":"86d40d715c9e00292242b351cfa76fd7997fa175bd66124f117ea7a4d81209c5","source":{"kind":"arxiv","id":"2106.02016","version":2},"attestation_state":"computed","paper":{"title":"Semantic-WER: A Unified Metric for the Evaluation of ASR Transcript for End Usability","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Somnath Roy","submitted_at":"2021-06-03T17:35:14Z","abstract_excerpt":"Recent advances in supervised, semi-supervised and self-supervised deep learning algorithms have shown significant improvement in the performance of automatic speech recognition(ASR) systems. The state-of-the-art systems have achieved a word error rate (WER) less than 5%. However, in the past, researchers have argued the non-suitability of the WER metric for the evaluation of ASR systems for downstream tasks such as spoken language understanding (SLU) and information retrieval. The reason is that the WER works at the surface level and does not include any syntactic and semantic knowledge.The c"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2106.02016","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2021-06-03T17:35:14Z","cross_cats_sorted":["cs.SD","eess.AS"],"title_canon_sha256":"39924197a14c30a194bf47621dd774d6c943a6c2a2f7202615eb7461a2790c29","abstract_canon_sha256":"8f5fa1c7dc25b6b83c9e8da400891914162f580a8bbc93e8b220f9191a7550db"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:23:11.114926Z","signature_b64":"8zhDbSFA2n2M9D4hoSyDvdIp4Zq1dClL9leMOMkNs6Ws4EKo9kTwClhO2ii7Rp6pVuA/GqyoMgaJk24X8alcCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"86d40d715c9e00292242b351cfa76fd7997fa175bd66124f117ea7a4d81209c5","last_reissued_at":"2026-07-05T03:23:11.114595Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:23:11.114595Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Semantic-WER: A Unified Metric for the Evaluation of ASR Transcript for End Usability","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Somnath Roy","submitted_at":"2021-06-03T17:35:14Z","abstract_excerpt":"Recent advances in supervised, semi-supervised and self-supervised deep learning algorithms have shown significant improvement in the performance of automatic speech recognition(ASR) systems. The state-of-the-art systems have achieved a word error rate (WER) less than 5%. However, in the past, researchers have argued the non-suitability of the WER metric for the evaluation of ASR systems for downstream tasks such as spoken language understanding (SLU) and information retrieval. The reason is that the WER works at the surface level and does not include any syntactic and semantic knowledge.The c"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2106.02016","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2106.02016/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2106.02016","created_at":"2026-07-05T03:23:11.114649+00:00"},{"alias_kind":"arxiv_version","alias_value":"2106.02016v2","created_at":"2026-07-05T03:23:11.114649+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2106.02016","created_at":"2026-07-05T03:23:11.114649+00:00"},{"alias_kind":"pith_short_12","alias_value":"Q3KA24K4TYAC","created_at":"2026-07-05T03:23:11.114649+00:00"},{"alias_kind":"pith_short_16","alias_value":"Q3KA24K4TYACSISC","created_at":"2026-07-05T03:23:11.114649+00:00"},{"alias_kind":"pith_short_8","alias_value":"Q3KA24K4","created_at":"2026-07-05T03:23:11.114649+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.29430","citing_title":"Towards Human-Like Interactive Speech Recognition With Agentic Correction and Semantic Evaluation","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21928","citing_title":"Evaluation of Automatic Speech Recognition Using Generative Large Language Models","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09121","citing_title":"Interactive ASR: Towards Human-Like Interaction and Semantic Coherence Evaluation for Agentic Speech Recognition","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Q3KA24K4TYACSISCWNI47J3P26","json":"https://pith.science/pith/Q3KA24K4TYACSISCWNI47J3P26.json","graph_json":"https://pith.science/api/pith-number/Q3KA24K4TYACSISCWNI47J3P26/graph.json","events_json":"https://pith.science/api/pith-number/Q3KA24K4TYACSISCWNI47J3P26/events.json","paper":"https://pith.science/paper/Q3KA24K4"},"agent_actions":{"view_html":"https://pith.science/pith/Q3KA24K4TYACSISCWNI47J3P26","download_json":"https://pith.science/pith/Q3KA24K4TYACSISCWNI47J3P26.json","view_paper":"https://pith.science/paper/Q3KA24K4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2106.02016&json=true","fetch_graph":"https://pith.science/api/pith-number/Q3KA24K4TYACSISCWNI47J3P26/graph.json","fetch_events":"https://pith.science/api/pith-number/Q3KA24K4TYACSISCWNI47J3P26/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Q3KA24K4TYACSISCWNI47J3P26/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Q3KA24K4TYACSISCWNI47J3P26/action/storage_attestation","attest_author":"https://pith.science/pith/Q3KA24K4TYACSISCWNI47J3P26/action/author_attestation","sign_citation":"https://pith.science/pith/Q3KA24K4TYACSISCWNI47J3P26/action/citation_signature","submit_replication":"https://pith.science/pith/Q3KA24K4TYACSISCWNI47J3P26/action/replication_record"}},"created_at":"2026-07-05T03:23:11.114649+00:00","updated_at":"2026-07-05T03:23:11.114649+00:00"}