{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:OHH2XYWRVH5OYDLIFPOSA6DY2Y","short_pith_number":"pith:OHH2XYWR","schema_version":"1.0","canonical_sha256":"71cfabe2d1a9faec0d682bdd207878d61528f8381dedd8849ee56c4b705791cf","source":{"kind":"arxiv","id":"2204.00175","version":2},"attestation_state":"computed","paper":{"title":"Alternate Intermediate Conditioning with Syllable-level and Character-level Targets for Japanese ASR","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Tatsuya Komatsu, Yusuke Fujita, Yusuke Kida","submitted_at":"2022-04-01T02:51:22Z","abstract_excerpt":"End-to-end automatic speech recognition directly maps input speech to characters. However, the mapping can be problematic when several different pronunciations should be mapped into one character or when one pronunciation is shared among many different characters. Japanese ASR suffers the most from such many-to-one and one-to-many mapping problems due to Japanese kanji characters. To alleviate the problems, we introduce explicit interaction between characters and syllables using Self-conditioned connectionist temporal classification (CTC), in which the upper layers are ``self-conditioned'' on "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2204.00175","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2022-04-01T02:51:22Z","cross_cats_sorted":["cs.SD","eess.AS"],"title_canon_sha256":"000238f9ff73e700f3d32f2d6de747e1a192633f906cce60645364201d5d9529","abstract_canon_sha256":"b05b45c79f71bfcedfccf17dc4e4141b3a346d187a455dfed6a45e9e08005d2e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:50:05.337674Z","signature_b64":"6ZO5WwctPznSM6PYRrnakkdQ4xfZiwwEGRqTDQLyIZmGZSXjHjoKjKam8/zHniKVLSVPzhbUHFjAP3hp3NODDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"71cfabe2d1a9faec0d682bdd207878d61528f8381dedd8849ee56c4b705791cf","last_reissued_at":"2026-07-05T05:50:05.337224Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:50:05.337224Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Alternate Intermediate Conditioning with Syllable-level and Character-level Targets for Japanese ASR","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.SD","eess.AS"],"primary_cat":"cs.CL","authors_text":"Tatsuya Komatsu, Yusuke Fujita, Yusuke Kida","submitted_at":"2022-04-01T02:51:22Z","abstract_excerpt":"End-to-end automatic speech recognition directly maps input speech to characters. However, the mapping can be problematic when several different pronunciations should be mapped into one character or when one pronunciation is shared among many different characters. Japanese ASR suffers the most from such many-to-one and one-to-many mapping problems due to Japanese kanji characters. To alleviate the problems, we introduce explicit interaction between characters and syllables using Self-conditioned connectionist temporal classification (CTC), in which the upper layers are ``self-conditioned'' on "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2204.00175","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2204.00175/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2204.00175","created_at":"2026-07-05T05:50:05.337289+00:00"},{"alias_kind":"arxiv_version","alias_value":"2204.00175v2","created_at":"2026-07-05T05:50:05.337289+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2204.00175","created_at":"2026-07-05T05:50:05.337289+00:00"},{"alias_kind":"pith_short_12","alias_value":"OHH2XYWRVH5O","created_at":"2026-07-05T05:50:05.337289+00:00"},{"alias_kind":"pith_short_16","alias_value":"OHH2XYWRVH5OYDLI","created_at":"2026-07-05T05:50:05.337289+00:00"},{"alias_kind":"pith_short_8","alias_value":"OHH2XYWR","created_at":"2026-07-05T05:50:05.337289+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.13079","citing_title":"Cross-modal Knowledge Transfer Learning as Graph Matching Based on Optimal Transport for ASR","ref_index":16,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OHH2XYWRVH5OYDLIFPOSA6DY2Y","json":"https://pith.science/pith/OHH2XYWRVH5OYDLIFPOSA6DY2Y.json","graph_json":"https://pith.science/api/pith-number/OHH2XYWRVH5OYDLIFPOSA6DY2Y/graph.json","events_json":"https://pith.science/api/pith-number/OHH2XYWRVH5OYDLIFPOSA6DY2Y/events.json","paper":"https://pith.science/paper/OHH2XYWR"},"agent_actions":{"view_html":"https://pith.science/pith/OHH2XYWRVH5OYDLIFPOSA6DY2Y","download_json":"https://pith.science/pith/OHH2XYWRVH5OYDLIFPOSA6DY2Y.json","view_paper":"https://pith.science/paper/OHH2XYWR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2204.00175&json=true","fetch_graph":"https://pith.science/api/pith-number/OHH2XYWRVH5OYDLIFPOSA6DY2Y/graph.json","fetch_events":"https://pith.science/api/pith-number/OHH2XYWRVH5OYDLIFPOSA6DY2Y/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OHH2XYWRVH5OYDLIFPOSA6DY2Y/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OHH2XYWRVH5OYDLIFPOSA6DY2Y/action/storage_attestation","attest_author":"https://pith.science/pith/OHH2XYWRVH5OYDLIFPOSA6DY2Y/action/author_attestation","sign_citation":"https://pith.science/pith/OHH2XYWRVH5OYDLIFPOSA6DY2Y/action/citation_signature","submit_replication":"https://pith.science/pith/OHH2XYWRVH5OYDLIFPOSA6DY2Y/action/replication_record"}},"created_at":"2026-07-05T05:50:05.337289+00:00","updated_at":"2026-07-05T05:50:05.337289+00:00"}