{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:G4IYX6JDZX3IULTCZRHMETN72T","short_pith_number":"pith:G4IYX6JD","schema_version":"1.0","canonical_sha256":"37118bf923cdf68a2e62cc4ec24dbfd4ca689a61b616c8a233cc3aec9fa125b3","source":{"kind":"arxiv","id":"1911.12893","version":1},"attestation_state":"computed","paper":{"title":"GitHub Typo Corpus: A Large-Scale Multilingual Dataset of Misspellings and Grammatical Errors","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Masato Hagiwara, Masato Mita","submitted_at":"2019-11-28T22:57:45Z","abstract_excerpt":"The lack of large-scale datasets has been a major hindrance to the development of NLP tasks such as spelling correction and grammatical error correction (GEC). As a complementary new resource for these tasks, we present the GitHub Typo Corpus, a large-scale, multilingual dataset of misspellings and grammatical errors along with their corrections harvested from GitHub, a large and popular platform for hosting and sharing git repositories. The dataset, which we have made publicly available, contains more than 350k edits and 65M characters in more than 15 languages, making it the largest dataset "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1911.12893","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2019-11-28T22:57:45Z","cross_cats_sorted":[],"title_canon_sha256":"8d9f2fe1c4bd546be4905805f5d99dc45f34518f1f89d30664f3a2ea0abc9c54","abstract_canon_sha256":"f22ad33f4b89f510df23563c576696429aa36cd0087c9c6b5d03589a89ec5807"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:22:50.013047Z","signature_b64":"plGJsu3OfadGjuD1brJRv6Isrdon5JuTzLCMDJR5gqEYNsdgdcmTIW5DLrwN8PD9N4yyxivsAxWdzEpZhoOqBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"37118bf923cdf68a2e62cc4ec24dbfd4ca689a61b616c8a233cc3aec9fa125b3","last_reissued_at":"2026-07-05T00:22:50.012591Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:22:50.012591Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GitHub Typo Corpus: A Large-Scale Multilingual Dataset of Misspellings and Grammatical Errors","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Masato Hagiwara, Masato Mita","submitted_at":"2019-11-28T22:57:45Z","abstract_excerpt":"The lack of large-scale datasets has been a major hindrance to the development of NLP tasks such as spelling correction and grammatical error correction (GEC). As a complementary new resource for these tasks, we present the GitHub Typo Corpus, a large-scale, multilingual dataset of misspellings and grammatical errors along with their corrections harvested from GitHub, a large and popular platform for hosting and sharing git repositories. The dataset, which we have made publicly available, contains more than 350k edits and 65M characters in more than 15 languages, making it the largest dataset "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1911.12893","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1911.12893/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1911.12893","created_at":"2026-07-05T00:22:50.012651+00:00"},{"alias_kind":"arxiv_version","alias_value":"1911.12893v1","created_at":"2026-07-05T00:22:50.012651+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1911.12893","created_at":"2026-07-05T00:22:50.012651+00:00"},{"alias_kind":"pith_short_12","alias_value":"G4IYX6JDZX3I","created_at":"2026-07-05T00:22:50.012651+00:00"},{"alias_kind":"pith_short_16","alias_value":"G4IYX6JDZX3IULTC","created_at":"2026-07-05T00:22:50.012651+00:00"},{"alias_kind":"pith_short_8","alias_value":"G4IYX6JD","created_at":"2026-07-05T00:22:50.012651+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2509.22202","citing_title":"Library Hallucinations in LLM-Generated Code: A Risk Analysis Grounded in Developer Queries","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2509.17314","citing_title":"Clotho: Measuring Task-Specific Pre-Generation Test Adequacy for LLM Inputs","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/G4IYX6JDZX3IULTCZRHMETN72T","json":"https://pith.science/pith/G4IYX6JDZX3IULTCZRHMETN72T.json","graph_json":"https://pith.science/api/pith-number/G4IYX6JDZX3IULTCZRHMETN72T/graph.json","events_json":"https://pith.science/api/pith-number/G4IYX6JDZX3IULTCZRHMETN72T/events.json","paper":"https://pith.science/paper/G4IYX6JD"},"agent_actions":{"view_html":"https://pith.science/pith/G4IYX6JDZX3IULTCZRHMETN72T","download_json":"https://pith.science/pith/G4IYX6JDZX3IULTCZRHMETN72T.json","view_paper":"https://pith.science/paper/G4IYX6JD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1911.12893&json=true","fetch_graph":"https://pith.science/api/pith-number/G4IYX6JDZX3IULTCZRHMETN72T/graph.json","fetch_events":"https://pith.science/api/pith-number/G4IYX6JDZX3IULTCZRHMETN72T/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/G4IYX6JDZX3IULTCZRHMETN72T/action/timestamp_anchor","attest_storage":"https://pith.science/pith/G4IYX6JDZX3IULTCZRHMETN72T/action/storage_attestation","attest_author":"https://pith.science/pith/G4IYX6JDZX3IULTCZRHMETN72T/action/author_attestation","sign_citation":"https://pith.science/pith/G4IYX6JDZX3IULTCZRHMETN72T/action/citation_signature","submit_replication":"https://pith.science/pith/G4IYX6JDZX3IULTCZRHMETN72T/action/replication_record"}},"created_at":"2026-07-05T00:22:50.012651+00:00","updated_at":"2026-07-05T00:22:50.012651+00:00"}