{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:P4T76VZFPI6DZE4F7H3SHPQWAP","short_pith_number":"pith:P4T76VZF","schema_version":"1.0","canonical_sha256":"7f27ff57257a3c3c9385f9f723be1603ecfc09787f5ed676de1954e3975b0224","source":{"kind":"arxiv","id":"2406.16536","version":2},"attestation_state":"computed","paper":{"title":"C-LLM: Learn to Check Chinese Spelling Errors Character by Character","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Fandong Meng, Jie Zhou, Kunting Li, Liang He, Yong Hu","submitted_at":"2024-06-24T11:16:31Z","abstract_excerpt":"Chinese Spell Checking (CSC) aims to detect and correct spelling errors in sentences. Despite Large Language Models (LLMs) exhibit robust capabilities and are widely applied in various tasks, their performance on CSC is often unsatisfactory. We find that LLMs fail to meet the Chinese character-level constraints of the CSC task, namely equal length and phonetic similarity, leading to a performance bottleneck. Further analysis reveal that this issue stems from the granularity of tokenization, as current mixed character-word tokenization struggles to satisfy these character-level constraints. To "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.16536","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-06-24T11:16:31Z","cross_cats_sorted":[],"title_canon_sha256":"6475dd5e7c0f481837a9dc9932acf99daf160e6c4dd7bd424b7b2a4a0379918f","abstract_canon_sha256":"3128e006235aad54209ede3587f643231aec9804a7ae080850a0596db48390e9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:26:28.590732Z","signature_b64":"coU3iDbflnx//fVKtK5f4wjQWmPDX/IOw5VLUGqF9pQG5g054PPdPL3f3n+Go2PHpaD0iZxDY3mZbmsf8lm0BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7f27ff57257a3c3c9385f9f723be1603ecfc09787f5ed676de1954e3975b0224","last_reissued_at":"2026-07-05T09:26:28.590250Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:26:28.590250Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"C-LLM: Learn to Check Chinese Spelling Errors Character by Character","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Fandong Meng, Jie Zhou, Kunting Li, Liang He, Yong Hu","submitted_at":"2024-06-24T11:16:31Z","abstract_excerpt":"Chinese Spell Checking (CSC) aims to detect and correct spelling errors in sentences. Despite Large Language Models (LLMs) exhibit robust capabilities and are widely applied in various tasks, their performance on CSC is often unsatisfactory. We find that LLMs fail to meet the Chinese character-level constraints of the CSC task, namely equal length and phonetic similarity, leading to a performance bottleneck. Further analysis reveal that this issue stems from the granularity of tokenization, as current mixed character-word tokenization struggles to satisfy these character-level constraints. To "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.16536","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.16536/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.16536","created_at":"2026-07-05T09:26:28.590308+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.16536v2","created_at":"2026-07-05T09:26:28.590308+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.16536","created_at":"2026-07-05T09:26:28.590308+00:00"},{"alias_kind":"pith_short_12","alias_value":"P4T76VZFPI6D","created_at":"2026-07-05T09:26:28.590308+00:00"},{"alias_kind":"pith_short_16","alias_value":"P4T76VZFPI6DZE4F","created_at":"2026-07-05T09:26:28.590308+00:00"},{"alias_kind":"pith_short_8","alias_value":"P4T76VZF","created_at":"2026-07-05T09:26:28.590308+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.06887","citing_title":"Mixture of Small and Large Models for Chinese Spelling Check","ref_index":14,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/P4T76VZFPI6DZE4F7H3SHPQWAP","json":"https://pith.science/pith/P4T76VZFPI6DZE4F7H3SHPQWAP.json","graph_json":"https://pith.science/api/pith-number/P4T76VZFPI6DZE4F7H3SHPQWAP/graph.json","events_json":"https://pith.science/api/pith-number/P4T76VZFPI6DZE4F7H3SHPQWAP/events.json","paper":"https://pith.science/paper/P4T76VZF"},"agent_actions":{"view_html":"https://pith.science/pith/P4T76VZFPI6DZE4F7H3SHPQWAP","download_json":"https://pith.science/pith/P4T76VZFPI6DZE4F7H3SHPQWAP.json","view_paper":"https://pith.science/paper/P4T76VZF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.16536&json=true","fetch_graph":"https://pith.science/api/pith-number/P4T76VZFPI6DZE4F7H3SHPQWAP/graph.json","fetch_events":"https://pith.science/api/pith-number/P4T76VZFPI6DZE4F7H3SHPQWAP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/P4T76VZFPI6DZE4F7H3SHPQWAP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/P4T76VZFPI6DZE4F7H3SHPQWAP/action/storage_attestation","attest_author":"https://pith.science/pith/P4T76VZFPI6DZE4F7H3SHPQWAP/action/author_attestation","sign_citation":"https://pith.science/pith/P4T76VZFPI6DZE4F7H3SHPQWAP/action/citation_signature","submit_replication":"https://pith.science/pith/P4T76VZFPI6DZE4F7H3SHPQWAP/action/replication_record"}},"created_at":"2026-07-05T09:26:28.590308+00:00","updated_at":"2026-07-05T09:26:28.590308+00:00"}