{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:YSTRYEIVCBG2QBKC5UI6JZKBSW","short_pith_number":"pith:YSTRYEIV","schema_version":"1.0","canonical_sha256":"c4a71c1115104da80542ed11e4e54195a5eba08a42d386e322d1d1c68c9efb3f","source":{"kind":"arxiv","id":"2303.13592","version":4},"attestation_state":"computed","paper":{"title":"Prompting Multilingual Large Language Models to Generate Code-Mixed Texts: The Case of South East Asian Languages","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Alham Fikri Aji, Arjun Subramonian, Genta Indra Winata, Holy Lovenia, Jan Christian Blaise Cruz, Jessica Zosa Forde, Lintang Sutawika, Long Phan, Rowena Garcia, Ruochen Zhang, Samuel Cahyawijaya, Skyler Wang, Thamar Solorio, Yin Lin Tan, Zheng-Xin Yong","submitted_at":"2023-03-23T18:16:30Z","abstract_excerpt":"While code-mixing is a common linguistic practice in many parts of the world, collecting high-quality and low-cost code-mixed data remains a challenge for natural language processing (NLP) research. The recent proliferation of Large Language Models (LLMs) compels one to ask: how capable are these systems in generating code-mixed data? In this paper, we explore prompting multilingual LLMs in a zero-shot manner to generate code-mixed data for seven languages in South East Asia (SEA), namely Indonesian, Malay, Chinese, Tagalog, Vietnamese, Tamil, and Singlish. We find that publicly available mult"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.13592","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-03-23T18:16:30Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"5ddd07b8dac728a9ea16ba20b536487c82c365ea54589b369848f1b9c217aae3","abstract_canon_sha256":"6b27c8fb62a3fbbc914ab2639d4b95cc64db58e7bfbadc2033c452e1a85e65e9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:49:59.770870Z","signature_b64":"e4nFX9PwkOxSlWbkuy83Y/k9Tun9S9yqOICpeIKh9el39KmQ8ah5/b+H0fi5dlZJrJ/fKZcZCiQJERlIMufhCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c4a71c1115104da80542ed11e4e54195a5eba08a42d386e322d1d1c68c9efb3f","last_reissued_at":"2026-07-05T06:49:59.770412Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:49:59.770412Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Prompting Multilingual Large Language Models to Generate Code-Mixed Texts: The Case of South East Asian Languages","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Alham Fikri Aji, Arjun Subramonian, Genta Indra Winata, Holy Lovenia, Jan Christian Blaise Cruz, Jessica Zosa Forde, Lintang Sutawika, Long Phan, Rowena Garcia, Ruochen Zhang, Samuel Cahyawijaya, Skyler Wang, Thamar Solorio, Yin Lin Tan, Zheng-Xin Yong","submitted_at":"2023-03-23T18:16:30Z","abstract_excerpt":"While code-mixing is a common linguistic practice in many parts of the world, collecting high-quality and low-cost code-mixed data remains a challenge for natural language processing (NLP) research. The recent proliferation of Large Language Models (LLMs) compels one to ask: how capable are these systems in generating code-mixed data? In this paper, we explore prompting multilingual LLMs in a zero-shot manner to generate code-mixed data for seven languages in South East Asia (SEA), namely Indonesian, Malay, Chinese, Tagalog, Vietnamese, Tamil, and Singlish. We find that publicly available mult"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.13592","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.13592/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.13592","created_at":"2026-07-05T06:49:59.770468+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.13592v4","created_at":"2026-07-05T06:49:59.770468+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.13592","created_at":"2026-07-05T06:49:59.770468+00:00"},{"alias_kind":"pith_short_12","alias_value":"YSTRYEIVCBG2","created_at":"2026-07-05T06:49:59.770468+00:00"},{"alias_kind":"pith_short_16","alias_value":"YSTRYEIVCBG2QBKC","created_at":"2026-07-05T06:49:59.770468+00:00"},{"alias_kind":"pith_short_8","alias_value":"YSTRYEIV","created_at":"2026-07-05T06:49:59.770468+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2310.02446","citing_title":"Low-Resource Languages Jailbreak GPT-4","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2602.11181","citing_title":"Code Mixologist : A Practitioner's Guide to Building Code-Mixed LLMs","ref_index":29,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YSTRYEIVCBG2QBKC5UI6JZKBSW","json":"https://pith.science/pith/YSTRYEIVCBG2QBKC5UI6JZKBSW.json","graph_json":"https://pith.science/api/pith-number/YSTRYEIVCBG2QBKC5UI6JZKBSW/graph.json","events_json":"https://pith.science/api/pith-number/YSTRYEIVCBG2QBKC5UI6JZKBSW/events.json","paper":"https://pith.science/paper/YSTRYEIV"},"agent_actions":{"view_html":"https://pith.science/pith/YSTRYEIVCBG2QBKC5UI6JZKBSW","download_json":"https://pith.science/pith/YSTRYEIVCBG2QBKC5UI6JZKBSW.json","view_paper":"https://pith.science/paper/YSTRYEIV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.13592&json=true","fetch_graph":"https://pith.science/api/pith-number/YSTRYEIVCBG2QBKC5UI6JZKBSW/graph.json","fetch_events":"https://pith.science/api/pith-number/YSTRYEIVCBG2QBKC5UI6JZKBSW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YSTRYEIVCBG2QBKC5UI6JZKBSW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YSTRYEIVCBG2QBKC5UI6JZKBSW/action/storage_attestation","attest_author":"https://pith.science/pith/YSTRYEIVCBG2QBKC5UI6JZKBSW/action/author_attestation","sign_citation":"https://pith.science/pith/YSTRYEIVCBG2QBKC5UI6JZKBSW/action/citation_signature","submit_replication":"https://pith.science/pith/YSTRYEIVCBG2QBKC5UI6JZKBSW/action/replication_record"}},"created_at":"2026-07-05T06:49:59.770468+00:00","updated_at":"2026-07-05T06:49:59.770468+00:00"}