{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:2SJQLOXTP4XFTV626NR6Q3MEZK","short_pith_number":"pith:2SJQLOXT","schema_version":"1.0","canonical_sha256":"d49305baf37f2e59d7daf363e86d84ca898a186072b75c793b16032631135afc","source":{"kind":"arxiv","id":"2411.11072","version":2},"attestation_state":"computed","paper":{"title":"Multilingual Large Language Models: A Systematic Survey","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ant\\'onio Branco, Deyi Xiong, Haoran Sun, Jiangcun Du, Leiyu Pan, Menglong Cui, Renren Jin, Shaolin Zhu, Shaoyang Xu, Supryadi","submitted_at":"2024-11-17T13:21:26Z","abstract_excerpt":"This paper provides a comprehensive survey of the latest research on multilingual large language models (MLLMs). MLLMs not only are able to understand and generate language across linguistic boundaries, but also represent an important advancement in artificial intelligence. We first discuss the architecture and pre-training objectives of MLLMs, highlighting the key components and methodologies that contribute to their multilingual capabilities. We then discuss the construction of multilingual pre-training and alignment datasets, underscoring the importance of data quality and diversity in enha"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.11072","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-11-17T13:21:26Z","cross_cats_sorted":[],"title_canon_sha256":"754bb8edf7e7f6416d9bd3875c44c37ef9ea0d8e4c259ba94cfce8dc06b5b486","abstract_canon_sha256":"c581437db1c7724a5c812cf984dbff80d9aebe20ec2de8460ef449fd8a74f30e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:37:12.177557Z","signature_b64":"/O/A31zeCcBd9zS204QtRUgnZiJVZQZCtfMK7t0cdwgOkNzLTcl0EcLmK6cfwmfRmtzuGI2QEkgIhA31OXHWDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d49305baf37f2e59d7daf363e86d84ca898a186072b75c793b16032631135afc","last_reissued_at":"2026-07-05T09:37:12.177102Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:37:12.177102Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Multilingual Large Language Models: A Systematic Survey","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ant\\'onio Branco, Deyi Xiong, Haoran Sun, Jiangcun Du, Leiyu Pan, Menglong Cui, Renren Jin, Shaolin Zhu, Shaoyang Xu, Supryadi","submitted_at":"2024-11-17T13:21:26Z","abstract_excerpt":"This paper provides a comprehensive survey of the latest research on multilingual large language models (MLLMs). MLLMs not only are able to understand and generate language across linguistic boundaries, but also represent an important advancement in artificial intelligence. We first discuss the architecture and pre-training objectives of MLLMs, highlighting the key components and methodologies that contribute to their multilingual capabilities. We then discuss the construction of multilingual pre-training and alignment datasets, underscoring the importance of data quality and diversity in enha"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.11072","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.11072/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.11072","created_at":"2026-07-05T09:37:12.177159+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.11072v2","created_at":"2026-07-05T09:37:12.177159+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.11072","created_at":"2026-07-05T09:37:12.177159+00:00"},{"alias_kind":"pith_short_12","alias_value":"2SJQLOXTP4XF","created_at":"2026-07-05T09:37:12.177159+00:00"},{"alias_kind":"pith_short_16","alias_value":"2SJQLOXTP4XFTV62","created_at":"2026-07-05T09:37:12.177159+00:00"},{"alias_kind":"pith_short_8","alias_value":"2SJQLOXT","created_at":"2026-07-05T09:37:12.177159+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25821","citing_title":"SARA: Unlocking Multilingual Knowledge in Mixture-of-Experts via Semantically Anchored Routing Alignment","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00664","citing_title":"YOMI-Bench: A Benchmark for Evaluating Kanji Reading and Phonological Understanding of LLMs for Japanese","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2505.17101","citing_title":"A quantitative analysis of semantic information in deep representations of text and images","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2507.09205","citing_title":"From Curated Data to Scalable Models: Continual Pre-training of Dense and MoE Large Language Models for Tibetan","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27550","citing_title":"APPSI-139: A Parallel Corpus of English Application Privacy Policy Summarization and Interpretation","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2SJQLOXTP4XFTV626NR6Q3MEZK","json":"https://pith.science/pith/2SJQLOXTP4XFTV626NR6Q3MEZK.json","graph_json":"https://pith.science/api/pith-number/2SJQLOXTP4XFTV626NR6Q3MEZK/graph.json","events_json":"https://pith.science/api/pith-number/2SJQLOXTP4XFTV626NR6Q3MEZK/events.json","paper":"https://pith.science/paper/2SJQLOXT"},"agent_actions":{"view_html":"https://pith.science/pith/2SJQLOXTP4XFTV626NR6Q3MEZK","download_json":"https://pith.science/pith/2SJQLOXTP4XFTV626NR6Q3MEZK.json","view_paper":"https://pith.science/paper/2SJQLOXT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.11072&json=true","fetch_graph":"https://pith.science/api/pith-number/2SJQLOXTP4XFTV626NR6Q3MEZK/graph.json","fetch_events":"https://pith.science/api/pith-number/2SJQLOXTP4XFTV626NR6Q3MEZK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2SJQLOXTP4XFTV626NR6Q3MEZK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2SJQLOXTP4XFTV626NR6Q3MEZK/action/storage_attestation","attest_author":"https://pith.science/pith/2SJQLOXTP4XFTV626NR6Q3MEZK/action/author_attestation","sign_citation":"https://pith.science/pith/2SJQLOXTP4XFTV626NR6Q3MEZK/action/citation_signature","submit_replication":"https://pith.science/pith/2SJQLOXTP4XFTV626NR6Q3MEZK/action/replication_record"}},"created_at":"2026-07-05T09:37:12.177159+00:00","updated_at":"2026-07-05T09:37:12.177159+00:00"}