{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:LP3JHS4FHL6AJ4BCQN6XO2IO7O","short_pith_number":"pith:LP3JHS4F","schema_version":"1.0","canonical_sha256":"5bf693cb853afc04f022837d77690efb9ceca9dc40e7ee896625203f425defe1","source":{"kind":"arxiv","id":"2406.17456","version":1},"attestation_state":"computed","paper":{"title":"Improving Grammatical Error Correction via Contextual Data Augmentation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Baoxin Wang, Dayong Wu, Qingfu Zhu, Wanxiang Che, Yijun Liu, Yixuan Wang","submitted_at":"2024-06-25T10:49:56Z","abstract_excerpt":"Nowadays, data augmentation through synthetic data has been widely used in the field of Grammatical Error Correction (GEC) to alleviate the problem of data scarcity. However, these synthetic data are mainly used in the pre-training phase rather than the data-limited fine-tuning phase due to inconsistent error distribution and noisy labels. In this paper, we propose a synthetic data construction method based on contextual augmentation, which can ensure an efficient augmentation of the original data with a more consistent error distribution. Specifically, we combine rule-based substitution with "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.17456","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-06-25T10:49:56Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"369dd9490b53e849b5aeae1b4449b22595c5510c0882cb75fde4e3c6ca6ac2b0","abstract_canon_sha256":"874d263a19a409d1f1f91db04e3df6f8b69f130cb2b49bedbe99abccd9a4749b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:36:35.220496Z","signature_b64":"dknJpt9/SMQiG8lG5sEguJOQxdNjSZ4LXXEm3F2eJLiK8VbDBjrNUK8LALZIMeeK/I1OfL1aT+EvRupj7LT/DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5bf693cb853afc04f022837d77690efb9ceca9dc40e7ee896625203f425defe1","last_reissued_at":"2026-07-05T08:36:35.220006Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:36:35.220006Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Improving Grammatical Error Correction via Contextual Data Augmentation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Baoxin Wang, Dayong Wu, Qingfu Zhu, Wanxiang Che, Yijun Liu, Yixuan Wang","submitted_at":"2024-06-25T10:49:56Z","abstract_excerpt":"Nowadays, data augmentation through synthetic data has been widely used in the field of Grammatical Error Correction (GEC) to alleviate the problem of data scarcity. However, these synthetic data are mainly used in the pre-training phase rather than the data-limited fine-tuning phase due to inconsistent error distribution and noisy labels. In this paper, we propose a synthetic data construction method based on contextual augmentation, which can ensure an efficient augmentation of the original data with a more consistent error distribution. Specifically, we combine rule-based substitution with "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.17456","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.17456/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.17456","created_at":"2026-07-05T08:36:35.220066+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.17456v1","created_at":"2026-07-05T08:36:35.220066+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.17456","created_at":"2026-07-05T08:36:35.220066+00:00"},{"alias_kind":"pith_short_12","alias_value":"LP3JHS4FHL6A","created_at":"2026-07-05T08:36:35.220066+00:00"},{"alias_kind":"pith_short_16","alias_value":"LP3JHS4FHL6AJ4BC","created_at":"2026-07-05T08:36:35.220066+00:00"},{"alias_kind":"pith_short_8","alias_value":"LP3JHS4F","created_at":"2026-07-05T08:36:35.220066+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.18780","citing_title":"Harnessing Rule-Based Reinforcement Learning for Enhanced Grammatical Error Correction","ref_index":44,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LP3JHS4FHL6AJ4BCQN6XO2IO7O","json":"https://pith.science/pith/LP3JHS4FHL6AJ4BCQN6XO2IO7O.json","graph_json":"https://pith.science/api/pith-number/LP3JHS4FHL6AJ4BCQN6XO2IO7O/graph.json","events_json":"https://pith.science/api/pith-number/LP3JHS4FHL6AJ4BCQN6XO2IO7O/events.json","paper":"https://pith.science/paper/LP3JHS4F"},"agent_actions":{"view_html":"https://pith.science/pith/LP3JHS4FHL6AJ4BCQN6XO2IO7O","download_json":"https://pith.science/pith/LP3JHS4FHL6AJ4BCQN6XO2IO7O.json","view_paper":"https://pith.science/paper/LP3JHS4F","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.17456&json=true","fetch_graph":"https://pith.science/api/pith-number/LP3JHS4FHL6AJ4BCQN6XO2IO7O/graph.json","fetch_events":"https://pith.science/api/pith-number/LP3JHS4FHL6AJ4BCQN6XO2IO7O/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LP3JHS4FHL6AJ4BCQN6XO2IO7O/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LP3JHS4FHL6AJ4BCQN6XO2IO7O/action/storage_attestation","attest_author":"https://pith.science/pith/LP3JHS4FHL6AJ4BCQN6XO2IO7O/action/author_attestation","sign_citation":"https://pith.science/pith/LP3JHS4FHL6AJ4BCQN6XO2IO7O/action/citation_signature","submit_replication":"https://pith.science/pith/LP3JHS4FHL6AJ4BCQN6XO2IO7O/action/replication_record"}},"created_at":"2026-07-05T08:36:35.220066+00:00","updated_at":"2026-07-05T08:36:35.220066+00:00"}