{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:CE4HLX7FXWHJUFPVVPHHZ2IN47","short_pith_number":"pith:CE4HLX7F","schema_version":"1.0","canonical_sha256":"113875dfe5bd8e9a15f5abce7ce90de7d4526b29f447f20a668724db60d1a35a","source":{"kind":"arxiv","id":"2010.03449","version":5},"attestation_state":"computed","paper":{"title":"Super-Human Performance in Online Low-latency Recognition of Conversational Speech","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Alex Waibel, Sebastian Stueker, Thai-Son Nguyen","submitted_at":"2020-10-07T14:41:32Z","abstract_excerpt":"Achieving super-human performance in recognizing human speech has been a goal for several decades, as researchers have worked on increasingly challenging tasks. In the 1990's it was discovered, that conversational speech between two humans turns out to be considerably more difficult than read speech as hesitations, disfluencies, false starts and sloppy articulation complicate acoustic processing and require robust handling of acoustic, lexical and language context, jointly. Early attempts with statistical models could only reach error rates over 50% and far from human performance (WER of aroun"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2010.03449","kind":"arxiv","version":5},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2020-10-07T14:41:32Z","cross_cats_sorted":[],"title_canon_sha256":"525090c1225a1cf60527a975192c42570bc3c1d0480dcb8b4e69b7ad491181b1","abstract_canon_sha256":"a653c959864fd7c45001d973a062d9371b41a57b9c4f7b7df03272ce3e72a2c1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:00:46.582110Z","signature_b64":"VZ+pObF78DsSbnstFr5lgTIO1hFvXxPoocdaVPyRc0U5+4Sq0v22if6MINlOydueUVldWGXALCrup6vY1kWuCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"113875dfe5bd8e9a15f5abce7ce90de7d4526b29f447f20a668724db60d1a35a","last_reissued_at":"2026-07-05T03:00:46.581725Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:00:46.581725Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Super-Human Performance in Online Low-latency Recognition of Conversational Speech","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Alex Waibel, Sebastian Stueker, Thai-Son Nguyen","submitted_at":"2020-10-07T14:41:32Z","abstract_excerpt":"Achieving super-human performance in recognizing human speech has been a goal for several decades, as researchers have worked on increasingly challenging tasks. In the 1990's it was discovered, that conversational speech between two humans turns out to be considerably more difficult than read speech as hesitations, disfluencies, false starts and sloppy articulation complicate acoustic processing and require robust handling of acoustic, lexical and language context, jointly. Early attempts with statistical models could only reach error rates over 50% and far from human performance (WER of aroun"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2010.03449","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2010.03449/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2010.03449","created_at":"2026-07-05T03:00:46.581781+00:00"},{"alias_kind":"arxiv_version","alias_value":"2010.03449v5","created_at":"2026-07-05T03:00:46.581781+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2010.03449","created_at":"2026-07-05T03:00:46.581781+00:00"},{"alias_kind":"pith_short_12","alias_value":"CE4HLX7FXWHJ","created_at":"2026-07-05T03:00:46.581781+00:00"},{"alias_kind":"pith_short_16","alias_value":"CE4HLX7FXWHJUFPV","created_at":"2026-07-05T03:00:46.581781+00:00"},{"alias_kind":"pith_short_8","alias_value":"CE4HLX7F","created_at":"2026-07-05T03:00:46.581781+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.02178","citing_title":"Cocktail-Party Audio-Visual Speech Recognition","ref_index":11,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CE4HLX7FXWHJUFPVVPHHZ2IN47","json":"https://pith.science/pith/CE4HLX7FXWHJUFPVVPHHZ2IN47.json","graph_json":"https://pith.science/api/pith-number/CE4HLX7FXWHJUFPVVPHHZ2IN47/graph.json","events_json":"https://pith.science/api/pith-number/CE4HLX7FXWHJUFPVVPHHZ2IN47/events.json","paper":"https://pith.science/paper/CE4HLX7F"},"agent_actions":{"view_html":"https://pith.science/pith/CE4HLX7FXWHJUFPVVPHHZ2IN47","download_json":"https://pith.science/pith/CE4HLX7FXWHJUFPVVPHHZ2IN47.json","view_paper":"https://pith.science/paper/CE4HLX7F","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2010.03449&json=true","fetch_graph":"https://pith.science/api/pith-number/CE4HLX7FXWHJUFPVVPHHZ2IN47/graph.json","fetch_events":"https://pith.science/api/pith-number/CE4HLX7FXWHJUFPVVPHHZ2IN47/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CE4HLX7FXWHJUFPVVPHHZ2IN47/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CE4HLX7FXWHJUFPVVPHHZ2IN47/action/storage_attestation","attest_author":"https://pith.science/pith/CE4HLX7FXWHJUFPVVPHHZ2IN47/action/author_attestation","sign_citation":"https://pith.science/pith/CE4HLX7FXWHJUFPVVPHHZ2IN47/action/citation_signature","submit_replication":"https://pith.science/pith/CE4HLX7FXWHJUFPVVPHHZ2IN47/action/replication_record"}},"created_at":"2026-07-05T03:00:46.581781+00:00","updated_at":"2026-07-05T03:00:46.581781+00:00"}