{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:MW4427RT6PRESMLJULFFAHEN5E","short_pith_number":"pith:MW4427RT","schema_version":"1.0","canonical_sha256":"65b9cd7e33f3e2493169a2ca501c8de93602c7277ba4fc1b653ca6926afe608c","source":{"kind":"arxiv","id":"2401.06034","version":6},"attestation_state":"computed","paper":{"title":"LinguAlchemy: Fusing Typological and Geographical Elements for Unseen Language Generalization","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alham Fikri Aji, Ayu Purwarianti, Genta Indra Winata, Muhammad Farid Adilazuarda, Samuel Cahyawijaya","submitted_at":"2024-01-11T16:48:00Z","abstract_excerpt":"Pretrained language models (PLMs) have become remarkably adept at task and language generalization. Nonetheless, they often fail when faced with unseen languages. In this work, we present LinguAlchemy, a regularization method that incorporates various linguistic information covering typological, geographical, and phylogenetic features to align PLMs representation to the corresponding linguistic information on each language. Our LinguAlchemy significantly improves the performance of mBERT and XLM-R on low-resource languages in multiple downstream tasks such as intent classification, news classi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.06034","kind":"arxiv","version":6},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-01-11T16:48:00Z","cross_cats_sorted":[],"title_canon_sha256":"3703bceb1af31e1969f26b882f3927254184f65eb70c219f4d446eac282b493a","abstract_canon_sha256":"e9706987213e60b0d2491209b42a70586b39fd45daac1de28dd614c841f1332c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:15:29.055903Z","signature_b64":"yslxJpa4UxRlTHE6F1pz8q5+UqpQyEIRLuqWkHl1y3ZtHcREhax/V00TyKDO7AXzD3LJsvYq75GBrzpqfcsJBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"65b9cd7e33f3e2493169a2ca501c8de93602c7277ba4fc1b653ca6926afe608c","last_reissued_at":"2026-07-05T09:15:29.055338Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:15:29.055338Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LinguAlchemy: Fusing Typological and Geographical Elements for Unseen Language Generalization","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alham Fikri Aji, Ayu Purwarianti, Genta Indra Winata, Muhammad Farid Adilazuarda, Samuel Cahyawijaya","submitted_at":"2024-01-11T16:48:00Z","abstract_excerpt":"Pretrained language models (PLMs) have become remarkably adept at task and language generalization. Nonetheless, they often fail when faced with unseen languages. In this work, we present LinguAlchemy, a regularization method that incorporates various linguistic information covering typological, geographical, and phylogenetic features to align PLMs representation to the corresponding linguistic information on each language. Our LinguAlchemy significantly improves the performance of mBERT and XLM-R on low-resource languages in multiple downstream tasks such as intent classification, news classi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.06034","kind":"arxiv","version":6},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.06034/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.06034","created_at":"2026-07-05T09:15:29.055396+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.06034v6","created_at":"2026-07-05T09:15:29.055396+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.06034","created_at":"2026-07-05T09:15:29.055396+00:00"},{"alias_kind":"pith_short_12","alias_value":"MW4427RT6PRE","created_at":"2026-07-05T09:15:29.055396+00:00"},{"alias_kind":"pith_short_16","alias_value":"MW4427RT6PRESMLJ","created_at":"2026-07-05T09:15:29.055396+00:00"},{"alias_kind":"pith_short_8","alias_value":"MW4427RT","created_at":"2026-07-05T09:15:29.055396+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2509.05060","citing_title":"Entropy2Vec: Crosslingual Language Modeling Entropy as End-to-End Learnable Language Representations","ref_index":2,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MW4427RT6PRESMLJULFFAHEN5E","json":"https://pith.science/pith/MW4427RT6PRESMLJULFFAHEN5E.json","graph_json":"https://pith.science/api/pith-number/MW4427RT6PRESMLJULFFAHEN5E/graph.json","events_json":"https://pith.science/api/pith-number/MW4427RT6PRESMLJULFFAHEN5E/events.json","paper":"https://pith.science/paper/MW4427RT"},"agent_actions":{"view_html":"https://pith.science/pith/MW4427RT6PRESMLJULFFAHEN5E","download_json":"https://pith.science/pith/MW4427RT6PRESMLJULFFAHEN5E.json","view_paper":"https://pith.science/paper/MW4427RT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.06034&json=true","fetch_graph":"https://pith.science/api/pith-number/MW4427RT6PRESMLJULFFAHEN5E/graph.json","fetch_events":"https://pith.science/api/pith-number/MW4427RT6PRESMLJULFFAHEN5E/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MW4427RT6PRESMLJULFFAHEN5E/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MW4427RT6PRESMLJULFFAHEN5E/action/storage_attestation","attest_author":"https://pith.science/pith/MW4427RT6PRESMLJULFFAHEN5E/action/author_attestation","sign_citation":"https://pith.science/pith/MW4427RT6PRESMLJULFFAHEN5E/action/citation_signature","submit_replication":"https://pith.science/pith/MW4427RT6PRESMLJULFFAHEN5E/action/replication_record"}},"created_at":"2026-07-05T09:15:29.055396+00:00","updated_at":"2026-07-05T09:15:29.055396+00:00"}