{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:AUN65UAQ5ALS7LHYRYJLCLU5H6","short_pith_number":"pith:AUN65UAQ","schema_version":"1.0","canonical_sha256":"051beed010e8172facf88e12b12e9d3f9d03c1796e65ab3e3821cd34391b447d","source":{"kind":"arxiv","id":"2404.19159","version":1},"attestation_state":"computed","paper":{"title":"What Drives Performance in Multilingual Language Models?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ameeta Agrawal, Sina Bagheri Nezhad","submitted_at":"2024-04-29T23:49:19Z","abstract_excerpt":"This study investigates the factors influencing the performance of multilingual large language models (MLLMs) across diverse languages. We study 6 MLLMs, including masked language models, autoregressive models, and instruction-tuned LLMs, on the SIB-200 dataset, a topic classification dataset encompassing 204 languages. Our analysis considers three scenarios: ALL languages, SEEN languages (present in the model's pretraining data), and UNSEEN languages (not present or documented in the model's pretraining data in any meaningful way). We examine the impact of factors such as pretraining data siz"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.19159","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-04-29T23:49:19Z","cross_cats_sorted":[],"title_canon_sha256":"c746e2af7d0a78c128a152de514ca08b13c4883a021a8cd1c4847655c0664056","abstract_canon_sha256":"3303df50869d2cde1c672de0ce10a6e3db913b3d12ce05b72e11bbfd80557703"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:45:48.081119Z","signature_b64":"QLKlhqdYw7RngRW8L9HQmqb/fKfT5P1YA83HhswHP5ZMRudMPx5Hx/ffv5S8Joq9krFvJ+Vih/DsvD9W5989Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"051beed010e8172facf88e12b12e9d3f9d03c1796e65ab3e3821cd34391b447d","last_reissued_at":"2026-07-05T09:45:48.080691Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:45:48.080691Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"What Drives Performance in Multilingual Language Models?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ameeta Agrawal, Sina Bagheri Nezhad","submitted_at":"2024-04-29T23:49:19Z","abstract_excerpt":"This study investigates the factors influencing the performance of multilingual large language models (MLLMs) across diverse languages. We study 6 MLLMs, including masked language models, autoregressive models, and instruction-tuned LLMs, on the SIB-200 dataset, a topic classification dataset encompassing 204 languages. Our analysis considers three scenarios: ALL languages, SEEN languages (present in the model's pretraining data), and UNSEEN languages (not present or documented in the model's pretraining data in any meaningful way). We examine the impact of factors such as pretraining data siz"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.19159","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.19159/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.19159","created_at":"2026-07-05T09:45:48.080747+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.19159v1","created_at":"2026-07-05T09:45:48.080747+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.19159","created_at":"2026-07-05T09:45:48.080747+00:00"},{"alias_kind":"pith_short_12","alias_value":"AUN65UAQ5ALS","created_at":"2026-07-05T09:45:48.080747+00:00"},{"alias_kind":"pith_short_16","alias_value":"AUN65UAQ5ALS7LHY","created_at":"2026-07-05T09:45:48.080747+00:00"},{"alias_kind":"pith_short_8","alias_value":"AUN65UAQ","created_at":"2026-07-05T09:45:48.080747+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.09331","citing_title":"Beyond English: The Impact of Prompt Translation Strategies across Languages and Tasks in Multilingual LLMs","ref_index":34,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AUN65UAQ5ALS7LHYRYJLCLU5H6","json":"https://pith.science/pith/AUN65UAQ5ALS7LHYRYJLCLU5H6.json","graph_json":"https://pith.science/api/pith-number/AUN65UAQ5ALS7LHYRYJLCLU5H6/graph.json","events_json":"https://pith.science/api/pith-number/AUN65UAQ5ALS7LHYRYJLCLU5H6/events.json","paper":"https://pith.science/paper/AUN65UAQ"},"agent_actions":{"view_html":"https://pith.science/pith/AUN65UAQ5ALS7LHYRYJLCLU5H6","download_json":"https://pith.science/pith/AUN65UAQ5ALS7LHYRYJLCLU5H6.json","view_paper":"https://pith.science/paper/AUN65UAQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.19159&json=true","fetch_graph":"https://pith.science/api/pith-number/AUN65UAQ5ALS7LHYRYJLCLU5H6/graph.json","fetch_events":"https://pith.science/api/pith-number/AUN65UAQ5ALS7LHYRYJLCLU5H6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AUN65UAQ5ALS7LHYRYJLCLU5H6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AUN65UAQ5ALS7LHYRYJLCLU5H6/action/storage_attestation","attest_author":"https://pith.science/pith/AUN65UAQ5ALS7LHYRYJLCLU5H6/action/author_attestation","sign_citation":"https://pith.science/pith/AUN65UAQ5ALS7LHYRYJLCLU5H6/action/citation_signature","submit_replication":"https://pith.science/pith/AUN65UAQ5ALS7LHYRYJLCLU5H6/action/replication_record"}},"created_at":"2026-07-05T09:45:48.080747+00:00","updated_at":"2026-07-05T09:45:48.080747+00:00"}