{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:PZW42237HG7ZZMEZF4BTW2DS3E","short_pith_number":"pith:PZW42237","schema_version":"1.0","canonical_sha256":"7e6dcd6b7f39bf9cb0992f033b6872d9100b0a5c3c9e0f6a036cdac19cfc23be","source":{"kind":"arxiv","id":"2410.13153","version":1},"attestation_state":"computed","paper":{"title":"Better to Ask in English: Evaluation of Large Language Models on English, Low-resource and Cross-Lingual Settings","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Imran Razzak, Krishno Dey, Md. Arid Hasan, Prerona Tarannum, Usman Naseem","submitted_at":"2024-10-17T02:12:30Z","abstract_excerpt":"Large Language Models (LLMs) are trained on massive amounts of data, enabling their application across diverse domains and tasks. Despite their remarkable performance, most LLMs are developed and evaluated primarily in English. Recently, a few multi-lingual LLMs have emerged, but their performance in low-resource languages, especially the most spoken languages in South Asia, is less explored. To address this gap, in this study, we evaluate LLMs such as GPT-4, Llama 2, and Gemini to analyze their effectiveness in English compared to other low-resource languages from South Asia (e.g., Bangla, Hi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.13153","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-17T02:12:30Z","cross_cats_sorted":[],"title_canon_sha256":"3e3df17d60b85f61ad2198b53b7d5442a2280a9ba167a5df84085e5c343f5ff3","abstract_canon_sha256":"1a7424a4a944dc8cc614d50b3fb8c52e548c10805a39f825b9d854447e094852"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:21:56.414692Z","signature_b64":"23E/X4UNB5JwQDz6SJ/3MbUMNLnKgs/fsqDH5L1+2D9BslKon1JsP9ZrLCOActOQdT97uJ2YYtz+W5cccHD6AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7e6dcd6b7f39bf9cb0992f033b6872d9100b0a5c3c9e0f6a036cdac19cfc23be","last_reissued_at":"2026-07-05T09:21:56.414247Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:21:56.414247Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Better to Ask in English: Evaluation of Large Language Models on English, Low-resource and Cross-Lingual Settings","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Imran Razzak, Krishno Dey, Md. Arid Hasan, Prerona Tarannum, Usman Naseem","submitted_at":"2024-10-17T02:12:30Z","abstract_excerpt":"Large Language Models (LLMs) are trained on massive amounts of data, enabling their application across diverse domains and tasks. Despite their remarkable performance, most LLMs are developed and evaluated primarily in English. Recently, a few multi-lingual LLMs have emerged, but their performance in low-resource languages, especially the most spoken languages in South Asia, is less explored. To address this gap, in this study, we evaluate LLMs such as GPT-4, Llama 2, and Gemini to analyze their effectiveness in English compared to other low-resource languages from South Asia (e.g., Bangla, Hi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.13153","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.13153/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.13153","created_at":"2026-07-05T09:21:56.414304+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.13153v1","created_at":"2026-07-05T09:21:56.414304+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.13153","created_at":"2026-07-05T09:21:56.414304+00:00"},{"alias_kind":"pith_short_12","alias_value":"PZW42237HG7Z","created_at":"2026-07-05T09:21:56.414304+00:00"},{"alias_kind":"pith_short_16","alias_value":"PZW42237HG7ZZMEZ","created_at":"2026-07-05T09:21:56.414304+00:00"},{"alias_kind":"pith_short_8","alias_value":"PZW42237","created_at":"2026-07-05T09:21:56.414304+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2507.20917","citing_title":"MediQAl: A French Medical Question Answering Dataset for Knowledge and Reasoning Evaluation","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PZW42237HG7ZZMEZF4BTW2DS3E","json":"https://pith.science/pith/PZW42237HG7ZZMEZF4BTW2DS3E.json","graph_json":"https://pith.science/api/pith-number/PZW42237HG7ZZMEZF4BTW2DS3E/graph.json","events_json":"https://pith.science/api/pith-number/PZW42237HG7ZZMEZF4BTW2DS3E/events.json","paper":"https://pith.science/paper/PZW42237"},"agent_actions":{"view_html":"https://pith.science/pith/PZW42237HG7ZZMEZF4BTW2DS3E","download_json":"https://pith.science/pith/PZW42237HG7ZZMEZF4BTW2DS3E.json","view_paper":"https://pith.science/paper/PZW42237","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.13153&json=true","fetch_graph":"https://pith.science/api/pith-number/PZW42237HG7ZZMEZF4BTW2DS3E/graph.json","fetch_events":"https://pith.science/api/pith-number/PZW42237HG7ZZMEZF4BTW2DS3E/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PZW42237HG7ZZMEZF4BTW2DS3E/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PZW42237HG7ZZMEZF4BTW2DS3E/action/storage_attestation","attest_author":"https://pith.science/pith/PZW42237HG7ZZMEZF4BTW2DS3E/action/author_attestation","sign_citation":"https://pith.science/pith/PZW42237HG7ZZMEZF4BTW2DS3E/action/citation_signature","submit_replication":"https://pith.science/pith/PZW42237HG7ZZMEZF4BTW2DS3E/action/replication_record"}},"created_at":"2026-07-05T09:21:56.414304+00:00","updated_at":"2026-07-05T09:21:56.414304+00:00"}