{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XXTNMKMBLX474OVG4BXLLMWK7S","short_pith_number":"pith:XXTNMKMB","schema_version":"1.0","canonical_sha256":"bde6d629815df9fe3aa6e06eb5b2cafcb2764fb24503c4c878d912116e6b8f71","source":{"kind":"arxiv","id":"2411.17679","version":5},"attestation_state":"computed","paper":{"title":"Enhancing Character-Level Understanding in LLMs through Token Internal Structure Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Conglin Liu, Fei Liu, Jian He, Quanwei Shen, Yuchi Liu, Yu Kuang, Zhiqiang Zhao, Zhu Xu, Zihan Zhang","submitted_at":"2024-11-26T18:44:39Z","abstract_excerpt":"Tokenization methods like Byte-Pair Encoding (BPE) enhance computational efficiency in large language models (LLMs) but often obscure internal character structures within tokens. This limitation hinders LLMs' ability to predict precise character positions, which is crucial in tasks like Chinese Spelling Correction (CSC) where identifying the positions of misspelled characters accelerates correction processes. We propose Token Internal Position Awareness (TIPA), a method that significantly improves models' ability to capture character positions within tokens by training them on reverse characte"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.17679","kind":"arxiv","version":5},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-11-26T18:44:39Z","cross_cats_sorted":[],"title_canon_sha256":"b4d6161edbc95b77cdf7f7ebf7ddcdb670a8ed2042be97e6cc15316e92f5953f","abstract_canon_sha256":"5cb784e2d925855c3e457f00a41489bdac6f06804a3a2c7d08122b5e63a3ad4d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:18:16.092551Z","signature_b64":"p9KBZIMPvULPDdVYho2eBX0z1nd9BxykhPdF6Mm8FIrhrLQAfsoD7mWrLxCn6E4I1s90Ex3VcwnqlgpwzSlvDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bde6d629815df9fe3aa6e06eb5b2cafcb2764fb24503c4c878d912116e6b8f71","last_reissued_at":"2026-07-05T11:18:16.092009Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:18:16.092009Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Enhancing Character-Level Understanding in LLMs through Token Internal Structure Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Conglin Liu, Fei Liu, Jian He, Quanwei Shen, Yuchi Liu, Yu Kuang, Zhiqiang Zhao, Zhu Xu, Zihan Zhang","submitted_at":"2024-11-26T18:44:39Z","abstract_excerpt":"Tokenization methods like Byte-Pair Encoding (BPE) enhance computational efficiency in large language models (LLMs) but often obscure internal character structures within tokens. This limitation hinders LLMs' ability to predict precise character positions, which is crucial in tasks like Chinese Spelling Correction (CSC) where identifying the positions of misspelled characters accelerates correction processes. We propose Token Internal Position Awareness (TIPA), a method that significantly improves models' ability to capture character positions within tokens by training them on reverse characte"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.17679","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.17679/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.17679","created_at":"2026-07-05T11:18:16.092076+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.17679v5","created_at":"2026-07-05T11:18:16.092076+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.17679","created_at":"2026-07-05T11:18:16.092076+00:00"},{"alias_kind":"pith_short_12","alias_value":"XXTNMKMBLX47","created_at":"2026-07-05T11:18:16.092076+00:00"},{"alias_kind":"pith_short_16","alias_value":"XXTNMKMBLX474OVG","created_at":"2026-07-05T11:18:16.092076+00:00"},{"alias_kind":"pith_short_8","alias_value":"XXTNMKMB","created_at":"2026-07-05T11:18:16.092076+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.10641","citing_title":"Spelling-out is not Straightforward: LLMs' Capability of Tokenization from Token to Characters","ref_index":30,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XXTNMKMBLX474OVG4BXLLMWK7S","json":"https://pith.science/pith/XXTNMKMBLX474OVG4BXLLMWK7S.json","graph_json":"https://pith.science/api/pith-number/XXTNMKMBLX474OVG4BXLLMWK7S/graph.json","events_json":"https://pith.science/api/pith-number/XXTNMKMBLX474OVG4BXLLMWK7S/events.json","paper":"https://pith.science/paper/XXTNMKMB"},"agent_actions":{"view_html":"https://pith.science/pith/XXTNMKMBLX474OVG4BXLLMWK7S","download_json":"https://pith.science/pith/XXTNMKMBLX474OVG4BXLLMWK7S.json","view_paper":"https://pith.science/paper/XXTNMKMB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.17679&json=true","fetch_graph":"https://pith.science/api/pith-number/XXTNMKMBLX474OVG4BXLLMWK7S/graph.json","fetch_events":"https://pith.science/api/pith-number/XXTNMKMBLX474OVG4BXLLMWK7S/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XXTNMKMBLX474OVG4BXLLMWK7S/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XXTNMKMBLX474OVG4BXLLMWK7S/action/storage_attestation","attest_author":"https://pith.science/pith/XXTNMKMBLX474OVG4BXLLMWK7S/action/author_attestation","sign_citation":"https://pith.science/pith/XXTNMKMBLX474OVG4BXLLMWK7S/action/citation_signature","submit_replication":"https://pith.science/pith/XXTNMKMBLX474OVG4BXLLMWK7S/action/replication_record"}},"created_at":"2026-07-05T11:18:16.092076+00:00","updated_at":"2026-07-05T11:18:16.092076+00:00"}