{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:DDSWI5X7SZE557MVXP6VC3PHNT","short_pith_number":"pith:DDSWI5X7","schema_version":"1.0","canonical_sha256":"18e56476ff9649defd95bbfd516de76cc07b0b86baf675544f039eb318f983c9","source":{"kind":"arxiv","id":"2108.01122","version":2},"attestation_state":"computed","paper":{"title":"Automatic recognition of suprasegmentals in speech","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jiahong Yuan, Kenneth Church, Mark Liberman, Neville Ryant, Xingyu Cai","submitted_at":"2021-08-02T18:47:59Z","abstract_excerpt":"This study reports our efforts to improve automatic recognition of suprasegmentals by fine-tuning wav2vec 2.0 with CTC, a method that has been successful in automatic speech recognition. We demonstrate that the method can improve the state-of-the-art on automatic recognition of syllables, tones, and pitch accents. Utilizing segmental information, by employing tonal finals or tonal syllables as recognition units, can significantly improve Mandarin tone recognition. Language models are helpful when tonal syllables are used as recognition units, but not helpful when tones are recognition units. F"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2108.01122","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2021-08-02T18:47:59Z","cross_cats_sorted":[],"title_canon_sha256":"7d9f7854c3f84b21ecf95912a41f661f89316754648796919d466e0a6765b966","abstract_canon_sha256":"50bad01b96ed67219116f4e5bfc0a886f7126d354f2b443ef6f6a7d5a7e44d83"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:03:09.380062Z","signature_b64":"Iv4LIUrg7t6VZg/JXgil/9mr9yDP7zqAxXKLgVuxseT7FJPr1J1jI2ianxXM6OhCpVJw490u7FJUPfn83ijbAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"18e56476ff9649defd95bbfd516de76cc07b0b86baf675544f039eb318f983c9","last_reissued_at":"2026-07-05T03:03:09.379517Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:03:09.379517Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Automatic recognition of suprasegmentals in speech","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jiahong Yuan, Kenneth Church, Mark Liberman, Neville Ryant, Xingyu Cai","submitted_at":"2021-08-02T18:47:59Z","abstract_excerpt":"This study reports our efforts to improve automatic recognition of suprasegmentals by fine-tuning wav2vec 2.0 with CTC, a method that has been successful in automatic speech recognition. We demonstrate that the method can improve the state-of-the-art on automatic recognition of syllables, tones, and pitch accents. Utilizing segmental information, by employing tonal finals or tonal syllables as recognition units, can significantly improve Mandarin tone recognition. Language models are helpful when tonal syllables are used as recognition units, but not helpful when tones are recognition units. F"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2108.01122","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2108.01122/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2108.01122","created_at":"2026-07-05T03:03:09.379575+00:00"},{"alias_kind":"arxiv_version","alias_value":"2108.01122v2","created_at":"2026-07-05T03:03:09.379575+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2108.01122","created_at":"2026-07-05T03:03:09.379575+00:00"},{"alias_kind":"pith_short_12","alias_value":"DDSWI5X7SZE5","created_at":"2026-07-05T03:03:09.379575+00:00"},{"alias_kind":"pith_short_16","alias_value":"DDSWI5X7SZE557MV","created_at":"2026-07-05T03:03:09.379575+00:00"},{"alias_kind":"pith_short_8","alias_value":"DDSWI5X7","created_at":"2026-07-05T03:03:09.379575+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.02584","citing_title":"Prosodic Structure Beyond Lexical Content: A Study of Self-Supervised Learning","ref_index":48,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DDSWI5X7SZE557MVXP6VC3PHNT","json":"https://pith.science/pith/DDSWI5X7SZE557MVXP6VC3PHNT.json","graph_json":"https://pith.science/api/pith-number/DDSWI5X7SZE557MVXP6VC3PHNT/graph.json","events_json":"https://pith.science/api/pith-number/DDSWI5X7SZE557MVXP6VC3PHNT/events.json","paper":"https://pith.science/paper/DDSWI5X7"},"agent_actions":{"view_html":"https://pith.science/pith/DDSWI5X7SZE557MVXP6VC3PHNT","download_json":"https://pith.science/pith/DDSWI5X7SZE557MVXP6VC3PHNT.json","view_paper":"https://pith.science/paper/DDSWI5X7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2108.01122&json=true","fetch_graph":"https://pith.science/api/pith-number/DDSWI5X7SZE557MVXP6VC3PHNT/graph.json","fetch_events":"https://pith.science/api/pith-number/DDSWI5X7SZE557MVXP6VC3PHNT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DDSWI5X7SZE557MVXP6VC3PHNT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DDSWI5X7SZE557MVXP6VC3PHNT/action/storage_attestation","attest_author":"https://pith.science/pith/DDSWI5X7SZE557MVXP6VC3PHNT/action/author_attestation","sign_citation":"https://pith.science/pith/DDSWI5X7SZE557MVXP6VC3PHNT/action/citation_signature","submit_replication":"https://pith.science/pith/DDSWI5X7SZE557MVXP6VC3PHNT/action/replication_record"}},"created_at":"2026-07-05T03:03:09.379575+00:00","updated_at":"2026-07-05T03:03:09.379575+00:00"}