{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:BCEEH353RJSCHX6JUPLEY2K4MD","short_pith_number":"pith:BCEEH353","schema_version":"1.0","canonical_sha256":"088843efbb8a6423dfc9a3d64c695c60d68fb2fecfa5e099b14590bcaafeca0f","source":{"kind":"arxiv","id":"2305.14288","version":2},"attestation_state":"computed","paper":{"title":"LLM-powered Data Augmentation for Enhanced Cross-lingual Performance","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alham Fikri Aji, Chenxi Whitehouse, Monojit Choudhury","submitted_at":"2023-05-23T17:33:27Z","abstract_excerpt":"This paper explores the potential of leveraging Large Language Models (LLMs) for data augmentation in multilingual commonsense reasoning datasets where the available training data is extremely limited. To achieve this, we utilise several LLMs, namely Dolly-v2, StableVicuna, ChatGPT, and GPT-4, to augment three datasets: XCOPA, XWinograd, and XStoryCloze. Subsequently, we evaluate the effectiveness of fine-tuning smaller multilingual models, mBERT and XLMR, using the synthesised data. We compare the performance of training with data generated in English and target languages, as well as translat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.14288","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-05-23T17:33:27Z","cross_cats_sorted":[],"title_canon_sha256":"338fe5d467c2ea40c3dad6c343ca0c37cbaf0f7f7d87c679abca52db4fcc1aa3","abstract_canon_sha256":"7818b531f38d77ce7ede31347d2ec731347f60a0649f5c660d9f7cfc34f25e3b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:03:25.396779Z","signature_b64":"gXsZ+Y6kESqe11a6B+XB9RdUDVw4xtlgJfQVDhRFnvrvo5a55F80BMYcpo4q/dhlco+jfh3PoQhPQ/UAX3M4Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"088843efbb8a6423dfc9a3d64c695c60d68fb2fecfa5e099b14590bcaafeca0f","last_reissued_at":"2026-07-05T07:03:25.396321Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:03:25.396321Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LLM-powered Data Augmentation for Enhanced Cross-lingual Performance","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alham Fikri Aji, Chenxi Whitehouse, Monojit Choudhury","submitted_at":"2023-05-23T17:33:27Z","abstract_excerpt":"This paper explores the potential of leveraging Large Language Models (LLMs) for data augmentation in multilingual commonsense reasoning datasets where the available training data is extremely limited. To achieve this, we utilise several LLMs, namely Dolly-v2, StableVicuna, ChatGPT, and GPT-4, to augment three datasets: XCOPA, XWinograd, and XStoryCloze. Subsequently, we evaluate the effectiveness of fine-tuning smaller multilingual models, mBERT and XLMR, using the synthesised data. We compare the performance of training with data generated in English and target languages, as well as translat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.14288","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.14288/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.14288","created_at":"2026-07-05T07:03:25.396374+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.14288v2","created_at":"2026-07-05T07:03:25.396374+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.14288","created_at":"2026-07-05T07:03:25.396374+00:00"},{"alias_kind":"pith_short_12","alias_value":"BCEEH353RJSC","created_at":"2026-07-05T07:03:25.396374+00:00"},{"alias_kind":"pith_short_16","alias_value":"BCEEH353RJSCHX6J","created_at":"2026-07-05T07:03:25.396374+00:00"},{"alias_kind":"pith_short_8","alias_value":"BCEEH353","created_at":"2026-07-05T07:03:25.396374+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2509.04139","citing_title":"Enhancing Technical Documents Retrieval for RAG","ref_index":7,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BCEEH353RJSCHX6JUPLEY2K4MD","json":"https://pith.science/pith/BCEEH353RJSCHX6JUPLEY2K4MD.json","graph_json":"https://pith.science/api/pith-number/BCEEH353RJSCHX6JUPLEY2K4MD/graph.json","events_json":"https://pith.science/api/pith-number/BCEEH353RJSCHX6JUPLEY2K4MD/events.json","paper":"https://pith.science/paper/BCEEH353"},"agent_actions":{"view_html":"https://pith.science/pith/BCEEH353RJSCHX6JUPLEY2K4MD","download_json":"https://pith.science/pith/BCEEH353RJSCHX6JUPLEY2K4MD.json","view_paper":"https://pith.science/paper/BCEEH353","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.14288&json=true","fetch_graph":"https://pith.science/api/pith-number/BCEEH353RJSCHX6JUPLEY2K4MD/graph.json","fetch_events":"https://pith.science/api/pith-number/BCEEH353RJSCHX6JUPLEY2K4MD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BCEEH353RJSCHX6JUPLEY2K4MD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BCEEH353RJSCHX6JUPLEY2K4MD/action/storage_attestation","attest_author":"https://pith.science/pith/BCEEH353RJSCHX6JUPLEY2K4MD/action/author_attestation","sign_citation":"https://pith.science/pith/BCEEH353RJSCHX6JUPLEY2K4MD/action/citation_signature","submit_replication":"https://pith.science/pith/BCEEH353RJSCHX6JUPLEY2K4MD/action/replication_record"}},"created_at":"2026-07-05T07:03:25.396374+00:00","updated_at":"2026-07-05T07:03:25.396374+00:00"}