{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:EM3GTVPIGH5I7EHMKM6LKVVKT7","short_pith_number":"pith:EM3GTVPI","schema_version":"1.0","canonical_sha256":"233669d5e831fa8f90ec533cb556aa9fcc2d21386cda6d5d0c8aec3707fea9bb","source":{"kind":"arxiv","id":"2210.10442","version":1},"attestation_state":"computed","paper":{"title":"Linguistic Rules-Based Corpus Generation for Native Chinese Grammatical Error Correction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ding Zhang, Haitao Zheng, Li Yangning, Qingyu Zhou, Rongyi Sun, Ruiyang Liu, Shirong Ma, Shulin Huang, Yinghui Li, Ying Shen, Yunbo Cao, Zhongli Li","submitted_at":"2022-10-19T10:20:39Z","abstract_excerpt":"Chinese Grammatical Error Correction (CGEC) is both a challenging NLP task and a common application in human daily life. Recently, many data-driven approaches are proposed for the development of CGEC research. However, there are two major limitations in the CGEC field: First, the lack of high-quality annotated training corpora prevents the performance of existing CGEC models from being significantly improved. Second, the grammatical errors in widely used test sets are not made by native Chinese speakers, resulting in a significant gap between the CGEC models and the real application. In this p"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2210.10442","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2022-10-19T10:20:39Z","cross_cats_sorted":[],"title_canon_sha256":"3988c51b8805cf37cb7283eef158ba4dc75e5998470468c73ead20300bec79f9","abstract_canon_sha256":"c88038ad7d9c70044fed321a130f0920851ea324d2652ecff2f4674c1bd982d4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:08:19.761102Z","signature_b64":"CHBNeid4zEsl4P/cjxfZBwTfGtPe5icVd828P3tWVXEUqMSf+sV+CPa1GHVLk2Tfs2T+IkLH2me+dQAWzuONBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"233669d5e831fa8f90ec533cb556aa9fcc2d21386cda6d5d0c8aec3707fea9bb","last_reissued_at":"2026-07-05T05:08:19.760677Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:08:19.760677Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Linguistic Rules-Based Corpus Generation for Native Chinese Grammatical Error Correction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ding Zhang, Haitao Zheng, Li Yangning, Qingyu Zhou, Rongyi Sun, Ruiyang Liu, Shirong Ma, Shulin Huang, Yinghui Li, Ying Shen, Yunbo Cao, Zhongli Li","submitted_at":"2022-10-19T10:20:39Z","abstract_excerpt":"Chinese Grammatical Error Correction (CGEC) is both a challenging NLP task and a common application in human daily life. Recently, many data-driven approaches are proposed for the development of CGEC research. However, there are two major limitations in the CGEC field: First, the lack of high-quality annotated training corpora prevents the performance of existing CGEC models from being significantly improved. Second, the grammatical errors in widely used test sets are not made by native Chinese speakers, resulting in a significant gap between the CGEC models and the real application. In this p"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2210.10442","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2210.10442/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2210.10442","created_at":"2026-07-05T05:08:19.760737+00:00"},{"alias_kind":"arxiv_version","alias_value":"2210.10442v1","created_at":"2026-07-05T05:08:19.760737+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2210.10442","created_at":"2026-07-05T05:08:19.760737+00:00"},{"alias_kind":"pith_short_12","alias_value":"EM3GTVPIGH5I","created_at":"2026-07-05T05:08:19.760737+00:00"},{"alias_kind":"pith_short_16","alias_value":"EM3GTVPIGH5I7EHM","created_at":"2026-07-05T05:08:19.760737+00:00"},{"alias_kind":"pith_short_8","alias_value":"EM3GTVPI","created_at":"2026-07-05T05:08:19.760737+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2501.00330","citing_title":"Exploring the Implicit Semantic Ability of Multimodal Large Language Models: A Pilot Study on Entity Set Expansion","ref_index":6,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EM3GTVPIGH5I7EHMKM6LKVVKT7","json":"https://pith.science/pith/EM3GTVPIGH5I7EHMKM6LKVVKT7.json","graph_json":"https://pith.science/api/pith-number/EM3GTVPIGH5I7EHMKM6LKVVKT7/graph.json","events_json":"https://pith.science/api/pith-number/EM3GTVPIGH5I7EHMKM6LKVVKT7/events.json","paper":"https://pith.science/paper/EM3GTVPI"},"agent_actions":{"view_html":"https://pith.science/pith/EM3GTVPIGH5I7EHMKM6LKVVKT7","download_json":"https://pith.science/pith/EM3GTVPIGH5I7EHMKM6LKVVKT7.json","view_paper":"https://pith.science/paper/EM3GTVPI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2210.10442&json=true","fetch_graph":"https://pith.science/api/pith-number/EM3GTVPIGH5I7EHMKM6LKVVKT7/graph.json","fetch_events":"https://pith.science/api/pith-number/EM3GTVPIGH5I7EHMKM6LKVVKT7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EM3GTVPIGH5I7EHMKM6LKVVKT7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EM3GTVPIGH5I7EHMKM6LKVVKT7/action/storage_attestation","attest_author":"https://pith.science/pith/EM3GTVPIGH5I7EHMKM6LKVVKT7/action/author_attestation","sign_citation":"https://pith.science/pith/EM3GTVPIGH5I7EHMKM6LKVVKT7/action/citation_signature","submit_replication":"https://pith.science/pith/EM3GTVPIGH5I7EHMKM6LKVVKT7/action/replication_record"}},"created_at":"2026-07-05T05:08:19.760737+00:00","updated_at":"2026-07-05T05:08:19.760737+00:00"}