{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:CQNSW7TOIP26BNT43GXMJNCM7O","short_pith_number":"pith:CQNSW7TO","schema_version":"1.0","canonical_sha256":"141b2b7e6e43f5e0b67cd9aec4b44cfbb4dbc10fd008ff2eacb1e5fcedaf2c6f","source":{"kind":"arxiv","id":"2603.29042","version":2},"attestation_state":"computed","paper":{"title":"An Empirical Recipe for Universal Phone Recognition","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Chin-Jou Li, David R. Mortensen, Eunjung Yeo, Kwanghee Choi, Shikhar Bharadwaj, Shinji Watanabe, William Chen","submitted_at":"2026-03-30T22:12:48Z","abstract_excerpt":"Phone recognition (PR) is a key enabler of multilingual and low-resource speech processing tasks, yet robust performance remains elusive. Highly performant English-focused models do not generalize across languages, while multilingual models underutilize pretrained representations. It also remains unclear how data scale, architecture, and training objective contribute to multilingual PR. We present PhoneticXEUS -- trained on large-scale multilingual data and achieving state-of-the-art performance on both multilingual (17.7% PFER) and accented English speech (10.6% PFER). Through controlled abla"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2603.29042","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2026-03-30T22:12:48Z","cross_cats_sorted":["cs.LG","cs.SD","eess.AS"],"title_canon_sha256":"c780bc625bdc0f5289850b6966efa1010b81b3bd766ea3b6d84ddc85554269eb","abstract_canon_sha256":"d514dcdfd28e8e2f3b836f301659d11ef14dea9f519812ef5c9ea3ccca4d9249"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-14T01:20:57.969036Z","signature_b64":"iJWaAo0MMX1rvAWTd7DLSLIVgMo91LQVvBJbEOLUEarAnq3XgD4bxuUZHPrGmSjeemICoCYh1sMXRd4Y3qHeCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"141b2b7e6e43f5e0b67cd9aec4b44cfbb4dbc10fd008ff2eacb1e5fcedaf2c6f","last_reissued_at":"2026-07-14T01:20:57.968105Z","signature_status":"signed_v1","first_computed_at":"2026-07-14T01:20:57.968105Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"An Empirical Recipe for Universal Phone Recognition","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Chin-Jou Li, David R. Mortensen, Eunjung Yeo, Kwanghee Choi, Shikhar Bharadwaj, Shinji Watanabe, William Chen","submitted_at":"2026-03-30T22:12:48Z","abstract_excerpt":"Phone recognition (PR) is a key enabler of multilingual and low-resource speech processing tasks, yet robust performance remains elusive. Highly performant English-focused models do not generalize across languages, while multilingual models underutilize pretrained representations. It also remains unclear how data scale, architecture, and training objective contribute to multilingual PR. We present PhoneticXEUS -- trained on large-scale multilingual data and achieving state-of-the-art performance on both multilingual (17.7% PFER) and accented English speech (10.6% PFER). Through controlled abla"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2603.29042","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2603.29042/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2603.29042","created_at":"2026-07-14T01:20:57.968552+00:00"},{"alias_kind":"arxiv_version","alias_value":"2603.29042v2","created_at":"2026-07-14T01:20:57.968552+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2603.29042","created_at":"2026-07-14T01:20:57.968552+00:00"},{"alias_kind":"pith_short_12","alias_value":"CQNSW7TOIP26","created_at":"2026-07-14T01:20:57.968552+00:00"},{"alias_kind":"pith_short_16","alias_value":"CQNSW7TOIP26BNT4","created_at":"2026-07-14T01:20:57.968552+00:00"},{"alias_kind":"pith_short_8","alias_value":"CQNSW7TO","created_at":"2026-07-14T01:20:57.968552+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2606.22824","citing_title":"BranchShine: Compact Raw-Audio-to-IPA Transcription with a RoPE E-Branchformer Encoder","ref_index":3,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CQNSW7TOIP26BNT43GXMJNCM7O","json":"https://pith.science/pith/CQNSW7TOIP26BNT43GXMJNCM7O.json","graph_json":"https://pith.science/api/pith-number/CQNSW7TOIP26BNT43GXMJNCM7O/graph.json","events_json":"https://pith.science/api/pith-number/CQNSW7TOIP26BNT43GXMJNCM7O/events.json","paper":"https://pith.science/paper/CQNSW7TO"},"agent_actions":{"view_html":"https://pith.science/pith/CQNSW7TOIP26BNT43GXMJNCM7O","download_json":"https://pith.science/pith/CQNSW7TOIP26BNT43GXMJNCM7O.json","view_paper":"https://pith.science/paper/CQNSW7TO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2603.29042&json=true","fetch_graph":"https://pith.science/api/pith-number/CQNSW7TOIP26BNT43GXMJNCM7O/graph.json","fetch_events":"https://pith.science/api/pith-number/CQNSW7TOIP26BNT43GXMJNCM7O/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CQNSW7TOIP26BNT43GXMJNCM7O/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CQNSW7TOIP26BNT43GXMJNCM7O/action/storage_attestation","attest_author":"https://pith.science/pith/CQNSW7TOIP26BNT43GXMJNCM7O/action/author_attestation","sign_citation":"https://pith.science/pith/CQNSW7TOIP26BNT43GXMJNCM7O/action/citation_signature","submit_replication":"https://pith.science/pith/CQNSW7TOIP26BNT43GXMJNCM7O/action/replication_record"}},"created_at":"2026-07-14T01:20:57.968552+00:00","updated_at":"2026-07-14T01:20:57.968552+00:00"}