{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:WBFJ7WZJ45VUBEZNGCZSLYRCPG","short_pith_number":"pith:WBFJ7WZJ","schema_version":"1.0","canonical_sha256":"b04a9fdb29e76b40932d30b325e22279b579d7dc56a9cb3330223f01ad514921","source":{"kind":"arxiv","id":"2404.06138","version":2},"attestation_state":"computed","paper":{"title":"Cendol: Open Instruction-tuned Generative Large Language Models for Indonesian Languages","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alham Fikri Aji, Ayu Purwarianti, Bryan Wilie, Dea Annisayanti Putri, Emmanuel Dave, Fajri Koto, Genta Indra Winata, Holy Lovenia, Jhonson Lee, Muhammad Ihza Mahendra, Nuur Shadieq, Pascale Fung, Rifki Afina Putri, Salsabil Maulana Akbar, Samuel Cahyawijaya, Wawan Cenggoro","submitted_at":"2024-04-09T09:04:30Z","abstract_excerpt":"Large language models (LLMs) show remarkable human-like capability in various domains and languages. However, a notable quality gap arises in low-resource languages, e.g., Indonesian indigenous languages, rendering them ineffective and inefficient in such linguistic contexts. To bridge this quality gap, we introduce Cendol, a collection of Indonesian LLMs encompassing both decoder-only and encoder-decoder architectures across a range of model sizes. We highlight Cendol's effectiveness across a diverse array of tasks, attaining 20% improvement, and demonstrate its capability to generalize to un"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.06138","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-04-09T09:04:30Z","cross_cats_sorted":[],"title_canon_sha256":"61c4fcd2a039aed35682acbe8e16ce1259b395c3d403a9c1e148936b489b6eb8","abstract_canon_sha256":"8dddb7cf6f98792b346206f38f69f21dbd34492110a55a447c532d390ad48ff4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:41:01.640900Z","signature_b64":"nZAVu9L7e4zMY6utGzOOrNtCvGdeTCzUuJS2vX3aPVy+htWY7aLzjNz0I1Jz6Pn8BXqROqTV2jsQ3ncbEUEEBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b04a9fdb29e76b40932d30b325e22279b579d7dc56a9cb3330223f01ad514921","last_reissued_at":"2026-07-05T08:41:01.640366Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:41:01.640366Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Cendol: Open Instruction-tuned Generative Large Language Models for Indonesian Languages","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alham Fikri Aji, Ayu Purwarianti, Bryan Wilie, Dea Annisayanti Putri, Emmanuel Dave, Fajri Koto, Genta Indra Winata, Holy Lovenia, Jhonson Lee, Muhammad Ihza Mahendra, Nuur Shadieq, Pascale Fung, Rifki Afina Putri, Salsabil Maulana Akbar, Samuel Cahyawijaya, Wawan Cenggoro","submitted_at":"2024-04-09T09:04:30Z","abstract_excerpt":"Large language models (LLMs) show remarkable human-like capability in various domains and languages. However, a notable quality gap arises in low-resource languages, e.g., Indonesian indigenous languages, rendering them ineffective and inefficient in such linguistic contexts. To bridge this quality gap, we introduce Cendol, a collection of Indonesian LLMs encompassing both decoder-only and encoder-decoder architectures across a range of model sizes. We highlight Cendol's effectiveness across a diverse array of tasks, attaining 20% improvement, and demonstrate its capability to generalize to un"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.06138","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.06138/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.06138","created_at":"2026-07-05T08:41:01.640432+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.06138v2","created_at":"2026-07-05T08:41:01.640432+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.06138","created_at":"2026-07-05T08:41:01.640432+00:00"},{"alias_kind":"pith_short_12","alias_value":"WBFJ7WZJ45VU","created_at":"2026-07-05T08:41:01.640432+00:00"},{"alias_kind":"pith_short_16","alias_value":"WBFJ7WZJ45VUBEZN","created_at":"2026-07-05T08:41:01.640432+00:00"},{"alias_kind":"pith_short_8","alias_value":"WBFJ7WZJ","created_at":"2026-07-05T08:41:01.640432+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.03619","citing_title":"Do Large Language Models Know Folktales? A Case Study of Yokai in Japanese Folktales","ref_index":2015,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WBFJ7WZJ45VUBEZNGCZSLYRCPG","json":"https://pith.science/pith/WBFJ7WZJ45VUBEZNGCZSLYRCPG.json","graph_json":"https://pith.science/api/pith-number/WBFJ7WZJ45VUBEZNGCZSLYRCPG/graph.json","events_json":"https://pith.science/api/pith-number/WBFJ7WZJ45VUBEZNGCZSLYRCPG/events.json","paper":"https://pith.science/paper/WBFJ7WZJ"},"agent_actions":{"view_html":"https://pith.science/pith/WBFJ7WZJ45VUBEZNGCZSLYRCPG","download_json":"https://pith.science/pith/WBFJ7WZJ45VUBEZNGCZSLYRCPG.json","view_paper":"https://pith.science/paper/WBFJ7WZJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.06138&json=true","fetch_graph":"https://pith.science/api/pith-number/WBFJ7WZJ45VUBEZNGCZSLYRCPG/graph.json","fetch_events":"https://pith.science/api/pith-number/WBFJ7WZJ45VUBEZNGCZSLYRCPG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WBFJ7WZJ45VUBEZNGCZSLYRCPG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WBFJ7WZJ45VUBEZNGCZSLYRCPG/action/storage_attestation","attest_author":"https://pith.science/pith/WBFJ7WZJ45VUBEZNGCZSLYRCPG/action/author_attestation","sign_citation":"https://pith.science/pith/WBFJ7WZJ45VUBEZNGCZSLYRCPG/action/citation_signature","submit_replication":"https://pith.science/pith/WBFJ7WZJ45VUBEZNGCZSLYRCPG/action/replication_record"}},"created_at":"2026-07-05T08:41:01.640432+00:00","updated_at":"2026-07-05T08:41:01.640432+00:00"}