{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:OUOQTVON7E3A76ZEBELL7JJ2SH","short_pith_number":"pith:OUOQTVON","schema_version":"1.0","canonical_sha256":"751d09d5cdf9360ffb240916bfa53a91f526a3ed4cedac676a59b933f972cdb7","source":{"kind":"arxiv","id":"2509.17912","version":2},"attestation_state":"computed","paper":{"title":"SiDiaC: Sinhala Diachronic Corpus","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Nevidu Jayatilleke, Nisansa de Silva","submitted_at":"2025-09-22T15:37:51Z","abstract_excerpt":"SiDiaC, the first comprehensive Sinhala Diachronic Corpus, covers a historical span from the 5th to the 20th century CE. SiDiaC comprises 58k words across 46 literary works, annotated carefully based on the written date, after filtering based on availability, authorship, copyright compliance, and data attribution. Texts from the National Library of Sri Lanka were digitised using Google Document AI OCR, followed by post-processing to correct formatting and modernise the orthography. The construction of SiDiaC was informed by practices from other corpora, such as FarPaHC, particularly in syntact"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2509.17912","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-09-22T15:37:51Z","cross_cats_sorted":[],"title_canon_sha256":"e4dd3121be27907b34b42ec969393e3b2134160ec5bad0a80ae94cbb2e040238","abstract_canon_sha256":"3715144d03dde95d072894648907d362bd13be047dee7f70b6fed9950487dbe1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-20T00:02:55.365746Z","signature_b64":"Yf/1R8+SmlKZcHt/tACyVvdyPLw/NSXizHDAQvHO0ePJ5sg2/WG3L3dDAfxgmV1MmSyIl6nOyoH59Ybb7RBbDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"751d09d5cdf9360ffb240916bfa53a91f526a3ed4cedac676a59b933f972cdb7","last_reissued_at":"2026-05-20T00:02:55.364847Z","signature_status":"signed_v1","first_computed_at":"2026-05-20T00:02:55.364847Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SiDiaC: Sinhala Diachronic Corpus","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Nevidu Jayatilleke, Nisansa de Silva","submitted_at":"2025-09-22T15:37:51Z","abstract_excerpt":"SiDiaC, the first comprehensive Sinhala Diachronic Corpus, covers a historical span from the 5th to the 20th century CE. SiDiaC comprises 58k words across 46 literary works, annotated carefully based on the written date, after filtering based on availability, authorship, copyright compliance, and data attribution. Texts from the National Library of Sri Lanka were digitised using Google Document AI OCR, followed by post-processing to correct formatting and modernise the orthography. The construction of SiDiaC was informed by practices from other corpora, such as FarPaHC, particularly in syntact"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.17912","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.17912/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2509.17912","created_at":"2026-05-20T00:02:55.364973+00:00"},{"alias_kind":"arxiv_version","alias_value":"2509.17912v2","created_at":"2026-05-20T00:02:55.364973+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.17912","created_at":"2026-05-20T00:02:55.364973+00:00"},{"alias_kind":"pith_short_12","alias_value":"OUOQTVON7E3A","created_at":"2026-05-20T00:02:55.364973+00:00"},{"alias_kind":"pith_short_16","alias_value":"OUOQTVON7E3A76ZE","created_at":"2026-05-20T00:02:55.364973+00:00"},{"alias_kind":"pith_short_8","alias_value":"OUOQTVON","created_at":"2026-05-20T00:02:55.364973+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OUOQTVON7E3A76ZEBELL7JJ2SH","json":"https://pith.science/pith/OUOQTVON7E3A76ZEBELL7JJ2SH.json","graph_json":"https://pith.science/api/pith-number/OUOQTVON7E3A76ZEBELL7JJ2SH/graph.json","events_json":"https://pith.science/api/pith-number/OUOQTVON7E3A76ZEBELL7JJ2SH/events.json","paper":"https://pith.science/paper/OUOQTVON"},"agent_actions":{"view_html":"https://pith.science/pith/OUOQTVON7E3A76ZEBELL7JJ2SH","download_json":"https://pith.science/pith/OUOQTVON7E3A76ZEBELL7JJ2SH.json","view_paper":"https://pith.science/paper/OUOQTVON","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2509.17912&json=true","fetch_graph":"https://pith.science/api/pith-number/OUOQTVON7E3A76ZEBELL7JJ2SH/graph.json","fetch_events":"https://pith.science/api/pith-number/OUOQTVON7E3A76ZEBELL7JJ2SH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OUOQTVON7E3A76ZEBELL7JJ2SH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OUOQTVON7E3A76ZEBELL7JJ2SH/action/storage_attestation","attest_author":"https://pith.science/pith/OUOQTVON7E3A76ZEBELL7JJ2SH/action/author_attestation","sign_citation":"https://pith.science/pith/OUOQTVON7E3A76ZEBELL7JJ2SH/action/citation_signature","submit_replication":"https://pith.science/pith/OUOQTVON7E3A76ZEBELL7JJ2SH/action/replication_record"}},"created_at":"2026-05-20T00:02:55.364973+00:00","updated_at":"2026-05-20T00:02:55.364973+00:00"}