{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:H6EWXZ2MUANA7P2AXSGDQIFIFE","short_pith_number":"pith:H6EWXZ2M","schema_version":"1.0","canonical_sha256":"3f896be74ca01a0fbf40bc8c3820a829174e8868a181267204be2fef633c442b","source":{"kind":"arxiv","id":"2407.02118","version":2},"attestation_state":"computed","paper":{"title":"Breaking Language Barriers: Cross-Lingual Continual Pre-Training at Scale","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Libo Qin, Li Yue, Ming Zhou, Wenbo Pan, Wenzhen Zheng, Xu Xu","submitted_at":"2024-07-02T10:06:41Z","abstract_excerpt":"In recent years, Large Language Models (LLMs) have made significant strides towards Artificial General Intelligence. However, training these models from scratch requires substantial computational resources and vast amounts of text data. In this paper, we explore an alternative approach to constructing an LLM for a new language by continually pretraining (CPT) from existing pretrained LLMs, instead of using randomly initialized parameters. Based on parallel experiments on 40 model sizes ranging from 40M to 5B parameters, we find that 1) CPT converges faster and saves significant resources in a "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.02118","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2024-07-02T10:06:41Z","cross_cats_sorted":[],"title_canon_sha256":"8d0235c1d7f3b3ff2c05929b3453650a759fd1276b893f12fd4b594450613b6d","abstract_canon_sha256":"39f967393ff6f0bd2338574b27ac32fbbc6fa77ef7525a3c61f2b267bf10216b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:14:35.244984Z","signature_b64":"C+At44DR88LVfWd76z/SKaYz7cEujq9S5VTuHVhgqb8v0y+a5ELrc7iO+gWp5A4kpOkXWSYk4lMgkzDlWg6ZCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3f896be74ca01a0fbf40bc8c3820a829174e8868a181267204be2fef633c442b","last_reissued_at":"2026-07-05T09:14:35.244458Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:14:35.244458Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Breaking Language Barriers: Cross-Lingual Continual Pre-Training at Scale","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Libo Qin, Li Yue, Ming Zhou, Wenbo Pan, Wenzhen Zheng, Xu Xu","submitted_at":"2024-07-02T10:06:41Z","abstract_excerpt":"In recent years, Large Language Models (LLMs) have made significant strides towards Artificial General Intelligence. However, training these models from scratch requires substantial computational resources and vast amounts of text data. In this paper, we explore an alternative approach to constructing an LLM for a new language by continually pretraining (CPT) from existing pretrained LLMs, instead of using randomly initialized parameters. Based on parallel experiments on 40 model sizes ranging from 40M to 5B parameters, we find that 1) CPT converges faster and saves significant resources in a "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.02118","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.02118/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.02118","created_at":"2026-07-05T09:14:35.244518+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.02118v2","created_at":"2026-07-05T09:14:35.244518+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.02118","created_at":"2026-07-05T09:14:35.244518+00:00"},{"alias_kind":"pith_short_12","alias_value":"H6EWXZ2MUANA","created_at":"2026-07-05T09:14:35.244518+00:00"},{"alias_kind":"pith_short_16","alias_value":"H6EWXZ2MUANA7P2A","created_at":"2026-07-05T09:14:35.244518+00:00"},{"alias_kind":"pith_short_8","alias_value":"H6EWXZ2M","created_at":"2026-07-05T09:14:35.244518+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.26601","citing_title":"FTibSuite: A Comprehensive Resource Suite for Tibetan Vision-Language Modeling","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/H6EWXZ2MUANA7P2AXSGDQIFIFE","json":"https://pith.science/pith/H6EWXZ2MUANA7P2AXSGDQIFIFE.json","graph_json":"https://pith.science/api/pith-number/H6EWXZ2MUANA7P2AXSGDQIFIFE/graph.json","events_json":"https://pith.science/api/pith-number/H6EWXZ2MUANA7P2AXSGDQIFIFE/events.json","paper":"https://pith.science/paper/H6EWXZ2M"},"agent_actions":{"view_html":"https://pith.science/pith/H6EWXZ2MUANA7P2AXSGDQIFIFE","download_json":"https://pith.science/pith/H6EWXZ2MUANA7P2AXSGDQIFIFE.json","view_paper":"https://pith.science/paper/H6EWXZ2M","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.02118&json=true","fetch_graph":"https://pith.science/api/pith-number/H6EWXZ2MUANA7P2AXSGDQIFIFE/graph.json","fetch_events":"https://pith.science/api/pith-number/H6EWXZ2MUANA7P2AXSGDQIFIFE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/H6EWXZ2MUANA7P2AXSGDQIFIFE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/H6EWXZ2MUANA7P2AXSGDQIFIFE/action/storage_attestation","attest_author":"https://pith.science/pith/H6EWXZ2MUANA7P2AXSGDQIFIFE/action/author_attestation","sign_citation":"https://pith.science/pith/H6EWXZ2MUANA7P2AXSGDQIFIFE/action/citation_signature","submit_replication":"https://pith.science/pith/H6EWXZ2MUANA7P2AXSGDQIFIFE/action/replication_record"}},"created_at":"2026-07-05T09:14:35.244518+00:00","updated_at":"2026-07-05T09:14:35.244518+00:00"}