{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:2XQESUJAQCBFEWYTGLK77HG44K","short_pith_number":"pith:2XQESUJA","schema_version":"1.0","canonical_sha256":"d5e04951208082525b1332d5ff9cdce28680f92873def19d13f8b93d67a2855a","source":{"kind":"arxiv","id":"2508.19089","version":1},"attestation_state":"computed","paper":{"title":"It's All About In-Context Learning! Teaching Extremely Low-Resource Languages to LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Carolina Scarton, Yue Li, Zhixue Zhao","submitted_at":"2025-08-26T14:51:10Z","abstract_excerpt":"Extremely low-resource languages, especially those written in rare scripts, as shown in Figure 1, remain largely unsupported by large language models (LLMs). This is due in part to compounding factors such as the lack of training data. This paper delivers the first comprehensive analysis of whether LLMs can acquire such languages purely via in-context learning (ICL), with or without auxiliary alignment signals, and how these methods compare to parameter-efficient fine-tuning (PEFT). We systematically evaluate 20 under-represented languages across three state-of-the-art multilingual LLMs. Our f"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.19089","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-08-26T14:51:10Z","cross_cats_sorted":[],"title_canon_sha256":"dcf833e84445f22ff0980afbd9dc020f6f0b02f04903f29799575362120b1777","abstract_canon_sha256":"2e9fad2e27e2614693b448bc16d9b7e99dd7618786c76609cdebdf35c297d16c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:59:36.530207Z","signature_b64":"a4d34b8NJpbDoH97bUujSeU8LLPL85MHcAEkb0bD68659C2VvC+XUbwno8F/PG4g/GuGA7CxkNTYDL0JR/5SAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d5e04951208082525b1332d5ff9cdce28680f92873def19d13f8b93d67a2855a","last_reissued_at":"2026-07-05T11:59:36.529794Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:59:36.529794Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"It's All About In-Context Learning! Teaching Extremely Low-Resource Languages to LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Carolina Scarton, Yue Li, Zhixue Zhao","submitted_at":"2025-08-26T14:51:10Z","abstract_excerpt":"Extremely low-resource languages, especially those written in rare scripts, as shown in Figure 1, remain largely unsupported by large language models (LLMs). This is due in part to compounding factors such as the lack of training data. This paper delivers the first comprehensive analysis of whether LLMs can acquire such languages purely via in-context learning (ICL), with or without auxiliary alignment signals, and how these methods compare to parameter-efficient fine-tuning (PEFT). We systematically evaluate 20 under-represented languages across three state-of-the-art multilingual LLMs. Our f"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.19089","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.19089/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.19089","created_at":"2026-07-05T11:59:36.529851+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.19089v1","created_at":"2026-07-05T11:59:36.529851+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.19089","created_at":"2026-07-05T11:59:36.529851+00:00"},{"alias_kind":"pith_short_12","alias_value":"2XQESUJAQCBF","created_at":"2026-07-05T11:59:36.529851+00:00"},{"alias_kind":"pith_short_16","alias_value":"2XQESUJAQCBFEWYT","created_at":"2026-07-05T11:59:36.529851+00:00"},{"alias_kind":"pith_short_8","alias_value":"2XQESUJA","created_at":"2026-07-05T11:59:36.529851+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11219","citing_title":"Afrispeech Semantics: Evaluating Audio Semantic Reasoning in Spoken Language Models Across Domains and Accents","ref_index":39,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2XQESUJAQCBFEWYTGLK77HG44K","json":"https://pith.science/pith/2XQESUJAQCBFEWYTGLK77HG44K.json","graph_json":"https://pith.science/api/pith-number/2XQESUJAQCBFEWYTGLK77HG44K/graph.json","events_json":"https://pith.science/api/pith-number/2XQESUJAQCBFEWYTGLK77HG44K/events.json","paper":"https://pith.science/paper/2XQESUJA"},"agent_actions":{"view_html":"https://pith.science/pith/2XQESUJAQCBFEWYTGLK77HG44K","download_json":"https://pith.science/pith/2XQESUJAQCBFEWYTGLK77HG44K.json","view_paper":"https://pith.science/paper/2XQESUJA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.19089&json=true","fetch_graph":"https://pith.science/api/pith-number/2XQESUJAQCBFEWYTGLK77HG44K/graph.json","fetch_events":"https://pith.science/api/pith-number/2XQESUJAQCBFEWYTGLK77HG44K/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2XQESUJAQCBFEWYTGLK77HG44K/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2XQESUJAQCBFEWYTGLK77HG44K/action/storage_attestation","attest_author":"https://pith.science/pith/2XQESUJAQCBFEWYTGLK77HG44K/action/author_attestation","sign_citation":"https://pith.science/pith/2XQESUJAQCBFEWYTGLK77HG44K/action/citation_signature","submit_replication":"https://pith.science/pith/2XQESUJAQCBFEWYTGLK77HG44K/action/replication_record"}},"created_at":"2026-07-05T11:59:36.529851+00:00","updated_at":"2026-07-05T11:59:36.529851+00:00"}