{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:ATFNCNHEYEUAPC3FXPCT2PYNGO","short_pith_number":"pith:ATFNCNHE","schema_version":"1.0","canonical_sha256":"04cad134e4c128078b65bbc53d3f0d33bf199fb302c996d8aeeffde71bfbd08d","source":{"kind":"arxiv","id":"2309.07445","version":3},"attestation_state":"computed","paper":{"title":"SIB-200: A Simple, Inclusive, and Big Evaluation Dataset for Topic Classification in 200+ Languages and Dialects","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Annie En-Shiun Lee, David Ifeoluwa Adelani, Hannah Liu, Haonan Gao, Jesujoba O. Alabi, Nikita Vassilyev, Xiaoyu Shen, Yanke Mao","submitted_at":"2023-09-14T05:56:49Z","abstract_excerpt":"Despite the progress we have recorded in the last few years in multilingual natural language processing, evaluation is typically limited to a small set of languages with available datasets which excludes a large number of low-resource languages. In this paper, we created SIB-200 -- a large-scale open-sourced benchmark dataset for topic classification in 200 languages and dialects to address the lack of evaluation dataset for Natural Language Understanding (NLU). For many of the languages covered in SIB-200, this is the first publicly available evaluation dataset for NLU. The dataset is based o"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.07445","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-09-14T05:56:49Z","cross_cats_sorted":[],"title_canon_sha256":"69d151335d4b51d4f5fbf96b8e34b5116b7119610831a758ccf0704307bebb78","abstract_canon_sha256":"f258c90c8e967603a22ab173929e5d4b219bb80e4ad1ca5d8c71f2eee15cb029"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:53:10.572647Z","signature_b64":"jtRK0cxdL991uTGszJSghfeZqIXaXQ1LJT/dlt32ix/mgIXAMrbch4Nyhd41dLrYqkfybv8gqGffgBTdZEEaCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"04cad134e4c128078b65bbc53d3f0d33bf199fb302c996d8aeeffde71bfbd08d","last_reissued_at":"2026-07-05T07:53:10.572147Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:53:10.572147Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SIB-200: A Simple, Inclusive, and Big Evaluation Dataset for Topic Classification in 200+ Languages and Dialects","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Annie En-Shiun Lee, David Ifeoluwa Adelani, Hannah Liu, Haonan Gao, Jesujoba O. Alabi, Nikita Vassilyev, Xiaoyu Shen, Yanke Mao","submitted_at":"2023-09-14T05:56:49Z","abstract_excerpt":"Despite the progress we have recorded in the last few years in multilingual natural language processing, evaluation is typically limited to a small set of languages with available datasets which excludes a large number of low-resource languages. In this paper, we created SIB-200 -- a large-scale open-sourced benchmark dataset for topic classification in 200 languages and dialects to address the lack of evaluation dataset for Natural Language Understanding (NLU). For many of the languages covered in SIB-200, this is the first publicly available evaluation dataset for NLU. The dataset is based o"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.07445","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.07445/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.07445","created_at":"2026-07-05T07:53:10.572206+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.07445v3","created_at":"2026-07-05T07:53:10.572206+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.07445","created_at":"2026-07-05T07:53:10.572206+00:00"},{"alias_kind":"pith_short_12","alias_value":"ATFNCNHEYEUA","created_at":"2026-07-05T07:53:10.572206+00:00"},{"alias_kind":"pith_short_16","alias_value":"ATFNCNHEYEUAPC3F","created_at":"2026-07-05T07:53:10.572206+00:00"},{"alias_kind":"pith_short_8","alias_value":"ATFNCNHE","created_at":"2026-07-05T07:53:10.572206+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24460","citing_title":"The African Language Tax: Quantifying the Cost, Latency, and Context Penalty of Tokenizing African Languages in Frontier LLMs","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24718","citing_title":"The Tokenizer Tax Across 25 European Languages: Domain Invariance, Cross-Lingual Few-Shot Effects, and the Ukrainian Penalty","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29738","citing_title":"Multi-Legal-Bench: Evaluating LLMs on Legal Reasoning Across Jurisdictions, Languages, and Legal Traditions","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2404.02534","citing_title":"ANGOFA: Leveraging OFA Embedding Initialization and Synthetic Data for Angolan Language Model","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2509.04471","citing_title":"MOSAIC: A Multilingual, Taxonomy-Agnostic, and Computationally Efficient Approach for Radiological Report Classification","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2507.12720","citing_title":"FLEXITOKENS: Flexible Tokenization for Evolving Language Models","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ATFNCNHEYEUAPC3FXPCT2PYNGO","json":"https://pith.science/pith/ATFNCNHEYEUAPC3FXPCT2PYNGO.json","graph_json":"https://pith.science/api/pith-number/ATFNCNHEYEUAPC3FXPCT2PYNGO/graph.json","events_json":"https://pith.science/api/pith-number/ATFNCNHEYEUAPC3FXPCT2PYNGO/events.json","paper":"https://pith.science/paper/ATFNCNHE"},"agent_actions":{"view_html":"https://pith.science/pith/ATFNCNHEYEUAPC3FXPCT2PYNGO","download_json":"https://pith.science/pith/ATFNCNHEYEUAPC3FXPCT2PYNGO.json","view_paper":"https://pith.science/paper/ATFNCNHE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.07445&json=true","fetch_graph":"https://pith.science/api/pith-number/ATFNCNHEYEUAPC3FXPCT2PYNGO/graph.json","fetch_events":"https://pith.science/api/pith-number/ATFNCNHEYEUAPC3FXPCT2PYNGO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ATFNCNHEYEUAPC3FXPCT2PYNGO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ATFNCNHEYEUAPC3FXPCT2PYNGO/action/storage_attestation","attest_author":"https://pith.science/pith/ATFNCNHEYEUAPC3FXPCT2PYNGO/action/author_attestation","sign_citation":"https://pith.science/pith/ATFNCNHEYEUAPC3FXPCT2PYNGO/action/citation_signature","submit_replication":"https://pith.science/pith/ATFNCNHEYEUAPC3FXPCT2PYNGO/action/replication_record"}},"created_at":"2026-07-05T07:53:10.572206+00:00","updated_at":"2026-07-05T07:53:10.572206+00:00"}