{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:4D4ZWAXBFOJONLFO4MER2JWAVV","short_pith_number":"pith:4D4ZWAXB","schema_version":"1.0","canonical_sha256":"e0f99b02e12b92e6acaee3091d26c0ad7e67f02a86d48efd97cf5fe5070d20dc","source":{"kind":"arxiv","id":"2506.04527","version":1},"attestation_state":"computed","paper":{"title":"Grapheme-Coherent Phonemic and Prosodic Annotation of Speech by Implicit and Explicit Grapheme Conditioning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","eess.AS"],"primary_cat":"cs.SD","authors_text":"Byeongseon Park, Hien Ohnaka, Ryuichi Yamamoto, Yuma Shirahata","submitted_at":"2025-06-05T00:24:00Z","abstract_excerpt":"We propose a model to obtain phonemic and prosodic labels of speech that are coherent with graphemes. Unlike previous methods that simply fine-tune a pre-trained ASR model with the labels, the proposed model conditions the label generation on corresponding graphemes by two methods: 1) Add implicit grapheme conditioning through prompt encoder using pre-trained BERT features. 2) Explicitly prune the label hypotheses inconsistent with the grapheme during inference. These methods enable obtaining parallel data of speech, the labels, and graphemes, which is applicable to various downstream tasks su"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.04527","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SD","submitted_at":"2025-06-05T00:24:00Z","cross_cats_sorted":["cs.CL","eess.AS"],"title_canon_sha256":"84106424b54f5970d4b5a8df4fa8d7832cd94dfeb53ce227e40b63366121fa6d","abstract_canon_sha256":"93d9ae8a7b2329a86b26d9594e67bae26fc9a3e169112b0329014d06c4cd7f2d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:16:15.948650Z","signature_b64":"Reg87H5yKEogxvpik+odzkoUaHJS6Fm1mN8WIilsnxmnmxpnMKU3u8B9TZJnoQBo1TkEYuLOiIgp6y5byhv1BA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e0f99b02e12b92e6acaee3091d26c0ad7e67f02a86d48efd97cf5fe5070d20dc","last_reissued_at":"2026-07-05T11:16:15.948082Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:16:15.948082Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Grapheme-Coherent Phonemic and Prosodic Annotation of Speech by Implicit and Explicit Grapheme Conditioning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","eess.AS"],"primary_cat":"cs.SD","authors_text":"Byeongseon Park, Hien Ohnaka, Ryuichi Yamamoto, Yuma Shirahata","submitted_at":"2025-06-05T00:24:00Z","abstract_excerpt":"We propose a model to obtain phonemic and prosodic labels of speech that are coherent with graphemes. Unlike previous methods that simply fine-tune a pre-trained ASR model with the labels, the proposed model conditions the label generation on corresponding graphemes by two methods: 1) Add implicit grapheme conditioning through prompt encoder using pre-trained BERT features. 2) Explicitly prune the label hypotheses inconsistent with the grapheme during inference. These methods enable obtaining parallel data of speech, the labels, and graphemes, which is applicable to various downstream tasks su"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.04527","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.04527/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.04527","created_at":"2026-07-05T11:16:15.948168+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.04527v1","created_at":"2026-07-05T11:16:15.948168+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.04527","created_at":"2026-07-05T11:16:15.948168+00:00"},{"alias_kind":"pith_short_12","alias_value":"4D4ZWAXBFOJO","created_at":"2026-07-05T11:16:15.948168+00:00"},{"alias_kind":"pith_short_16","alias_value":"4D4ZWAXBFOJONLFO","created_at":"2026-07-05T11:16:15.948168+00:00"},{"alias_kind":"pith_short_8","alias_value":"4D4ZWAXB","created_at":"2026-07-05T11:16:15.948168+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.04527","citing_title":"Grapheme-Coherent Phonemic and Prosodic Annotation of Speech by Implicit and Explicit Grapheme Conditioning","ref_index":2,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4D4ZWAXBFOJONLFO4MER2JWAVV","json":"https://pith.science/pith/4D4ZWAXBFOJONLFO4MER2JWAVV.json","graph_json":"https://pith.science/api/pith-number/4D4ZWAXBFOJONLFO4MER2JWAVV/graph.json","events_json":"https://pith.science/api/pith-number/4D4ZWAXBFOJONLFO4MER2JWAVV/events.json","paper":"https://pith.science/paper/4D4ZWAXB"},"agent_actions":{"view_html":"https://pith.science/pith/4D4ZWAXBFOJONLFO4MER2JWAVV","download_json":"https://pith.science/pith/4D4ZWAXBFOJONLFO4MER2JWAVV.json","view_paper":"https://pith.science/paper/4D4ZWAXB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.04527&json=true","fetch_graph":"https://pith.science/api/pith-number/4D4ZWAXBFOJONLFO4MER2JWAVV/graph.json","fetch_events":"https://pith.science/api/pith-number/4D4ZWAXBFOJONLFO4MER2JWAVV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4D4ZWAXBFOJONLFO4MER2JWAVV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4D4ZWAXBFOJONLFO4MER2JWAVV/action/storage_attestation","attest_author":"https://pith.science/pith/4D4ZWAXBFOJONLFO4MER2JWAVV/action/author_attestation","sign_citation":"https://pith.science/pith/4D4ZWAXBFOJONLFO4MER2JWAVV/action/citation_signature","submit_replication":"https://pith.science/pith/4D4ZWAXBFOJONLFO4MER2JWAVV/action/replication_record"}},"created_at":"2026-07-05T11:16:15.948168+00:00","updated_at":"2026-07-05T11:16:15.948168+00:00"}