{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ZRNW4YSS7CZXEZEUG2BYTBW6Q3","short_pith_number":"pith:ZRNW4YSS","schema_version":"1.0","canonical_sha256":"cc5b6e6252f8b372649436838986de86c2f13c4746771905e5fda1f314bd35ad","source":{"kind":"arxiv","id":"2509.03162","version":1},"attestation_state":"computed","paper":{"title":"SinhalaMMLU: A Comprehensive Benchmark for Evaluating Multitask Language Understanding in Sinhala","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ashmari Pramodya, Chamila Liyanage, Heshan Shalinda, Hidetaka Kamigaito, Nirasha Nelki, Randil Pushpananda, Ruvan Weerasinghe, Taro Watanabe, Yusuke Sakai","submitted_at":"2025-09-03T09:22:39Z","abstract_excerpt":"Large Language Models (LLMs) demonstrate impressive general knowledge and reasoning abilities, yet their evaluation has predominantly focused on global or anglocentric subjects, often neglecting low-resource languages and culturally specific content. While recent multilingual benchmarks attempt to bridge this gap, many rely on automatic translation, which can introduce errors and misrepresent the original cultural context. To address this, we introduce SinhalaMMLU, the first multiple-choice question answering benchmark designed specifically for Sinhala, a low-resource language. The dataset inc"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2509.03162","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-09-03T09:22:39Z","cross_cats_sorted":[],"title_canon_sha256":"5d5addc30ed4a14b405b817026a90154dfa235aec5a3b23256c9db84e61e3425","abstract_canon_sha256":"289f85aec1727316499589f5b2b9216b65ab3a493ce7e2063d37175e4a647616"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:04:05.287213Z","signature_b64":"3BVltGpSDnwzERraTgnQpUfyNN8mJjaXlNdwCag50K09UOuV7PYrHrwI2AVxZbbcGYryz/WA/iKhg9+C+WopDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cc5b6e6252f8b372649436838986de86c2f13c4746771905e5fda1f314bd35ad","last_reissued_at":"2026-07-05T12:04:05.286714Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:04:05.286714Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SinhalaMMLU: A Comprehensive Benchmark for Evaluating Multitask Language Understanding in Sinhala","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ashmari Pramodya, Chamila Liyanage, Heshan Shalinda, Hidetaka Kamigaito, Nirasha Nelki, Randil Pushpananda, Ruvan Weerasinghe, Taro Watanabe, Yusuke Sakai","submitted_at":"2025-09-03T09:22:39Z","abstract_excerpt":"Large Language Models (LLMs) demonstrate impressive general knowledge and reasoning abilities, yet their evaluation has predominantly focused on global or anglocentric subjects, often neglecting low-resource languages and culturally specific content. While recent multilingual benchmarks attempt to bridge this gap, many rely on automatic translation, which can introduce errors and misrepresent the original cultural context. To address this, we introduce SinhalaMMLU, the first multiple-choice question answering benchmark designed specifically for Sinhala, a low-resource language. The dataset inc"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.03162","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.03162/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2509.03162","created_at":"2026-07-05T12:04:05.286772+00:00"},{"alias_kind":"arxiv_version","alias_value":"2509.03162v1","created_at":"2026-07-05T12:04:05.286772+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.03162","created_at":"2026-07-05T12:04:05.286772+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZRNW4YSS7CZX","created_at":"2026-07-05T12:04:05.286772+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZRNW4YSS7CZXEZEU","created_at":"2026-07-05T12:04:05.286772+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZRNW4YSS","created_at":"2026-07-05T12:04:05.286772+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZRNW4YSS7CZXEZEUG2BYTBW6Q3","json":"https://pith.science/pith/ZRNW4YSS7CZXEZEUG2BYTBW6Q3.json","graph_json":"https://pith.science/api/pith-number/ZRNW4YSS7CZXEZEUG2BYTBW6Q3/graph.json","events_json":"https://pith.science/api/pith-number/ZRNW4YSS7CZXEZEUG2BYTBW6Q3/events.json","paper":"https://pith.science/paper/ZRNW4YSS"},"agent_actions":{"view_html":"https://pith.science/pith/ZRNW4YSS7CZXEZEUG2BYTBW6Q3","download_json":"https://pith.science/pith/ZRNW4YSS7CZXEZEUG2BYTBW6Q3.json","view_paper":"https://pith.science/paper/ZRNW4YSS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2509.03162&json=true","fetch_graph":"https://pith.science/api/pith-number/ZRNW4YSS7CZXEZEUG2BYTBW6Q3/graph.json","fetch_events":"https://pith.science/api/pith-number/ZRNW4YSS7CZXEZEUG2BYTBW6Q3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZRNW4YSS7CZXEZEUG2BYTBW6Q3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZRNW4YSS7CZXEZEUG2BYTBW6Q3/action/storage_attestation","attest_author":"https://pith.science/pith/ZRNW4YSS7CZXEZEUG2BYTBW6Q3/action/author_attestation","sign_citation":"https://pith.science/pith/ZRNW4YSS7CZXEZEUG2BYTBW6Q3/action/citation_signature","submit_replication":"https://pith.science/pith/ZRNW4YSS7CZXEZEUG2BYTBW6Q3/action/replication_record"}},"created_at":"2026-07-05T12:04:05.286772+00:00","updated_at":"2026-07-05T12:04:05.286772+00:00"}