{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:K44E2NI6NXOET432PH6P2ERLGT","short_pith_number":"pith:K44E2NI6","schema_version":"1.0","canonical_sha256":"57384d351e6ddc49f37a79fcfd122b34c7a2afd445be565f3319dee5c644132d","source":{"kind":"arxiv","id":"2312.12683","version":2},"attestation_state":"computed","paper":{"title":"Turning English-centric LLMs Into Polyglots: How Much Multilinguality Is Needed?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Florian Schottmann, Rico Sennrich, Tannon Kew","submitted_at":"2023-12-20T00:49:52Z","abstract_excerpt":"The vast majority of today's large language models (LLMs) are English-centric, having been pretrained predominantly on English text. Yet, in order to meet user expectations, models need to be able to respond appropriately in multiple languages once deployed in downstream applications. This requires strong cross-lingual transfer abilities. In this work, we investigate the minimal amount of multilinguality required during finetuning to elicit cross-lingual generalisation in English-centric LLMs. In experiments across four LLMs, we find that multilingual instruction tuning with as few as two to t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.12683","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-12-20T00:49:52Z","cross_cats_sorted":[],"title_canon_sha256":"1dd9e677a7174ed799765c593f528bd5fc7cd4de2c742948b69f06bd91d8ec30","abstract_canon_sha256":"30180da65309e4f2ecbf45ab917a6e68cfefcc424a9f15dd5191baa33afef639"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:15:07.497604Z","signature_b64":"Xtzsk4D9/fRTzajEbvi3zFKWTd6UlByIsQaiV6vd93Xi07xxMF5Jvgg3o05HARMroYNyzKZM/ihHkd5+NZyNDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"57384d351e6ddc49f37a79fcfd122b34c7a2afd445be565f3319dee5c644132d","last_reissued_at":"2026-07-05T09:15:07.497137Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:15:07.497137Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Turning English-centric LLMs Into Polyglots: How Much Multilinguality Is Needed?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Florian Schottmann, Rico Sennrich, Tannon Kew","submitted_at":"2023-12-20T00:49:52Z","abstract_excerpt":"The vast majority of today's large language models (LLMs) are English-centric, having been pretrained predominantly on English text. Yet, in order to meet user expectations, models need to be able to respond appropriately in multiple languages once deployed in downstream applications. This requires strong cross-lingual transfer abilities. In this work, we investigate the minimal amount of multilinguality required during finetuning to elicit cross-lingual generalisation in English-centric LLMs. In experiments across four LLMs, we find that multilingual instruction tuning with as few as two to t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.12683","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.12683/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.12683","created_at":"2026-07-05T09:15:07.497195+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.12683v2","created_at":"2026-07-05T09:15:07.497195+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.12683","created_at":"2026-07-05T09:15:07.497195+00:00"},{"alias_kind":"pith_short_12","alias_value":"K44E2NI6NXOE","created_at":"2026-07-05T09:15:07.497195+00:00"},{"alias_kind":"pith_short_16","alias_value":"K44E2NI6NXOET432","created_at":"2026-07-05T09:15:07.497195+00:00"},{"alias_kind":"pith_short_8","alias_value":"K44E2NI6","created_at":"2026-07-05T09:15:07.497195+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2501.05122","citing_title":"Centurio: On Drivers of Multilingual Ability of Large Vision-Language Model","ref_index":6,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/K44E2NI6NXOET432PH6P2ERLGT","json":"https://pith.science/pith/K44E2NI6NXOET432PH6P2ERLGT.json","graph_json":"https://pith.science/api/pith-number/K44E2NI6NXOET432PH6P2ERLGT/graph.json","events_json":"https://pith.science/api/pith-number/K44E2NI6NXOET432PH6P2ERLGT/events.json","paper":"https://pith.science/paper/K44E2NI6"},"agent_actions":{"view_html":"https://pith.science/pith/K44E2NI6NXOET432PH6P2ERLGT","download_json":"https://pith.science/pith/K44E2NI6NXOET432PH6P2ERLGT.json","view_paper":"https://pith.science/paper/K44E2NI6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.12683&json=true","fetch_graph":"https://pith.science/api/pith-number/K44E2NI6NXOET432PH6P2ERLGT/graph.json","fetch_events":"https://pith.science/api/pith-number/K44E2NI6NXOET432PH6P2ERLGT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/K44E2NI6NXOET432PH6P2ERLGT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/K44E2NI6NXOET432PH6P2ERLGT/action/storage_attestation","attest_author":"https://pith.science/pith/K44E2NI6NXOET432PH6P2ERLGT/action/author_attestation","sign_citation":"https://pith.science/pith/K44E2NI6NXOET432PH6P2ERLGT/action/citation_signature","submit_replication":"https://pith.science/pith/K44E2NI6NXOET432PH6P2ERLGT/action/replication_record"}},"created_at":"2026-07-05T09:15:07.497195+00:00","updated_at":"2026-07-05T09:15:07.497195+00:00"}