{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:IV37MAHEH7HS3H7SI5EM637DGD","short_pith_number":"pith:IV37MAHE","schema_version":"1.0","canonical_sha256":"4577f600e43fcf2d9ff24748cf6fe330e9ebe083f657c73cc69b6fee3c599dc6","source":{"kind":"arxiv","id":"2309.09400","version":1},"attestation_state":"computed","paper":{"title":"CulturaX: A Cleaned, Enormous, and Multilingual Dataset for Large Language Models in 167 Languages","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Chien Van Nguyen, Franck Dernoncourt, Hieu Man, Nghia Trung Ngo, Ryan A. Rossi, Thien Huu Nguyen, Thuat Nguyen, Viet Dac Lai","submitted_at":"2023-09-17T23:49:10Z","abstract_excerpt":"The driving factors behind the development of large language models (LLMs) with impressive learning capabilities are their colossal model sizes and extensive training datasets. Along with the progress in natural language processing, LLMs have been frequently made accessible to the public to foster deeper investigation and applications. However, when it comes to training datasets for these LLMs, especially the recent state-of-the-art models, they are often not fully disclosed. Creating training data for high-performing LLMs involves extensive cleaning and deduplication to ensure the necessary l"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.09400","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-09-17T23:49:10Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"5028988ebcf28c87c6b99c0393ab2183e3b7f35eff49522799fa6d1d805d315b","abstract_canon_sha256":"7a542cc37a3fc9e825a33272d0ac6ce7e075f5db7847a91a68f978915986adc6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:51:38.420749Z","signature_b64":"X51YJIJXoPXblJ21eyCpWgtHvIx7i6NKM/WAfTj7CEL074dfcH0mJapai9aMVdowytTtzt/DJ4l9FKD3oUEhAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4577f600e43fcf2d9ff24748cf6fe330e9ebe083f657c73cc69b6fee3c599dc6","last_reissued_at":"2026-07-05T06:51:38.420251Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:51:38.420251Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CulturaX: A Cleaned, Enormous, and Multilingual Dataset for Large Language Models in 167 Languages","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Chien Van Nguyen, Franck Dernoncourt, Hieu Man, Nghia Trung Ngo, Ryan A. Rossi, Thien Huu Nguyen, Thuat Nguyen, Viet Dac Lai","submitted_at":"2023-09-17T23:49:10Z","abstract_excerpt":"The driving factors behind the development of large language models (LLMs) with impressive learning capabilities are their colossal model sizes and extensive training datasets. Along with the progress in natural language processing, LLMs have been frequently made accessible to the public to foster deeper investigation and applications. However, when it comes to training datasets for these LLMs, especially the recent state-of-the-art models, they are often not fully disclosed. Creating training data for high-performing LLMs involves extensive cleaning and deduplication to ensure the necessary l"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.09400","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.09400/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.09400","created_at":"2026-07-05T06:51:38.420309+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.09400v1","created_at":"2026-07-05T06:51:38.420309+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.09400","created_at":"2026-07-05T06:51:38.420309+00:00"},{"alias_kind":"pith_short_12","alias_value":"IV37MAHEH7HS","created_at":"2026-07-05T06:51:38.420309+00:00"},{"alias_kind":"pith_short_16","alias_value":"IV37MAHEH7HS3H7S","created_at":"2026-07-05T06:51:38.420309+00:00"},{"alias_kind":"pith_short_8","alias_value":"IV37MAHE","created_at":"2026-07-05T06:51:38.420309+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.27019","citing_title":"MinGram: A Minimalist Unigram Tokenizer with High Compression and Competitive Morphological Alignment","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20993","citing_title":"Phonemes to the Rescue: Multilingual Tokenization Based on International Phonetic Alphabet","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24718","citing_title":"The Tokenizer Tax Across 25 European Languages: Domain Invariance, Cross-Lingual Few-Shot Effects, and the Ukrainian Penalty","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18083","citing_title":"A Data-Efficient Path to Multilingual LLMs: Language Expansion via Post-training PARAM$\\Delta$ Integration into Upcycled MoE","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15053","citing_title":"TFGN: Task-Free, Replay-Free Continual Pre-Training Without Catastrophic Forgetting at LLM Scale","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13429","citing_title":"TokAlign++: Advancing Vocabulary Adaptation via Better Token Alignment","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2403.04652","citing_title":"Yi: Open Foundation Models by 01.AI","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10590","citing_title":"Bridging Linguistic Gaps: Cross-Lingual Mapping in Pre-Training and Dataset for Enhanced Multilingual LLM Performance","ref_index":31,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IV37MAHEH7HS3H7SI5EM637DGD","json":"https://pith.science/pith/IV37MAHEH7HS3H7SI5EM637DGD.json","graph_json":"https://pith.science/api/pith-number/IV37MAHEH7HS3H7SI5EM637DGD/graph.json","events_json":"https://pith.science/api/pith-number/IV37MAHEH7HS3H7SI5EM637DGD/events.json","paper":"https://pith.science/paper/IV37MAHE"},"agent_actions":{"view_html":"https://pith.science/pith/IV37MAHEH7HS3H7SI5EM637DGD","download_json":"https://pith.science/pith/IV37MAHEH7HS3H7SI5EM637DGD.json","view_paper":"https://pith.science/paper/IV37MAHE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.09400&json=true","fetch_graph":"https://pith.science/api/pith-number/IV37MAHEH7HS3H7SI5EM637DGD/graph.json","fetch_events":"https://pith.science/api/pith-number/IV37MAHEH7HS3H7SI5EM637DGD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IV37MAHEH7HS3H7SI5EM637DGD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IV37MAHEH7HS3H7SI5EM637DGD/action/storage_attestation","attest_author":"https://pith.science/pith/IV37MAHEH7HS3H7SI5EM637DGD/action/author_attestation","sign_citation":"https://pith.science/pith/IV37MAHEH7HS3H7SI5EM637DGD/action/citation_signature","submit_replication":"https://pith.science/pith/IV37MAHEH7HS3H7SI5EM637DGD/action/replication_record"}},"created_at":"2026-07-05T06:51:38.420309+00:00","updated_at":"2026-07-05T06:51:38.420309+00:00"}