{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:FX523G5YIAUG4XN46FEH2KPU44","short_pith_number":"pith:FX523G5Y","schema_version":"1.0","canonical_sha256":"2dfbad9bb840286e5dbcf1487d29f4e72fa368af9bf036d16fcfa0a8a834a046","source":{"kind":"arxiv","id":"2506.04079","version":2},"attestation_state":"computed","paper":{"title":"EuroLLM-9B: Technical Report","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Alexandra Birch, Amin Farajian, Andr\\'e F. T. Martins, Barry Haddow, Duarte M. Alves, Fran\\c{c}ois Yvon, Jo\\~ao Alves, Jos\\'e G. C. de Souza, Jos\\'e Pombal, Manuel Faysse, Mateusz Klimaszewski, Nicolas Boizard, Nuno M. Guerreiro, Patrick Fernandes, Pedro Henrique Martins, Pierre Colombo, Ricardo Rei","submitted_at":"2025-06-04T15:43:31Z","abstract_excerpt":"This report presents EuroLLM-9B, a large language model trained from scratch to support the needs of European citizens by covering all 24 official European Union languages and 11 additional languages. EuroLLM addresses the issue of European languages being underrepresented and underserved in existing open large language models. We provide a comprehensive overview of EuroLLM-9B's development, including tokenizer design, architectural specifications, data filtering, and training procedures. We describe the pre-training data collection and filtering pipeline, including the creation of EuroFilter,"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.04079","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-04T15:43:31Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"6fbbaa7c9c8ba90baa6b13ae5b3e2a4cffa90ac4ed073b647b57a46b06c72813","abstract_canon_sha256":"d97fb0386ac9b316241f475a4d19d2a401dcdaa215dad61db9a955e6abb34bae"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:22:41.469353Z","signature_b64":"T9bwpMShWritAZaRuH3o4IrOyySj43+yZEp/tytN7TI5GM3vgnvHJDnzJILuGkZ+o07ajwrB37pSufcjfr+oDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2dfbad9bb840286e5dbcf1487d29f4e72fa368af9bf036d16fcfa0a8a834a046","last_reissued_at":"2026-07-05T11:22:41.468860Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:22:41.468860Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"EuroLLM-9B: Technical Report","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Alexandra Birch, Amin Farajian, Andr\\'e F. T. Martins, Barry Haddow, Duarte M. Alves, Fran\\c{c}ois Yvon, Jo\\~ao Alves, Jos\\'e G. C. de Souza, Jos\\'e Pombal, Manuel Faysse, Mateusz Klimaszewski, Nicolas Boizard, Nuno M. Guerreiro, Patrick Fernandes, Pedro Henrique Martins, Pierre Colombo, Ricardo Rei","submitted_at":"2025-06-04T15:43:31Z","abstract_excerpt":"This report presents EuroLLM-9B, a large language model trained from scratch to support the needs of European citizens by covering all 24 official European Union languages and 11 additional languages. EuroLLM addresses the issue of European languages being underrepresented and underserved in existing open large language models. We provide a comprehensive overview of EuroLLM-9B's development, including tokenizer design, architectural specifications, data filtering, and training procedures. We describe the pre-training data collection and filtering pipeline, including the creation of EuroFilter,"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.04079","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.04079/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.04079","created_at":"2026-07-05T11:22:41.468916+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.04079v2","created_at":"2026-07-05T11:22:41.468916+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.04079","created_at":"2026-07-05T11:22:41.468916+00:00"},{"alias_kind":"pith_short_12","alias_value":"FX523G5YIAUG","created_at":"2026-07-05T11:22:41.468916+00:00"},{"alias_kind":"pith_short_16","alias_value":"FX523G5YIAUG4XN4","created_at":"2026-07-05T11:22:41.468916+00:00"},{"alias_kind":"pith_short_8","alias_value":"FX523G5Y","created_at":"2026-07-05T11:22:41.468916+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08731","citing_title":"Trusting sovereign language models as scientific instruments: evidence from Portugal's AMALIA","ref_index":7,"is_internal_anchor":true},{"citing_arxiv_id":"2606.21203","citing_title":"When Context Misleads: Surprisal, Energy and Attention Entropy as Metrics of Coherence Illusions in LLMs","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01995","citing_title":"CARTE: A Benchmark for Mapping Language Model Knowledge Across France","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29476","citing_title":"Comparative Evaluation of Machine Translation Systems on Images with Text","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23721","citing_title":"Is a Document Educational or Just Wikipedia-Style? -- Pitfalls of Classifier-Based Quality Filtering","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11290","citing_title":"Polyglot Teachers: Evaluating Language Models for Multilingual Synthetic Data Generation","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FX523G5YIAUG4XN46FEH2KPU44","json":"https://pith.science/pith/FX523G5YIAUG4XN46FEH2KPU44.json","graph_json":"https://pith.science/api/pith-number/FX523G5YIAUG4XN46FEH2KPU44/graph.json","events_json":"https://pith.science/api/pith-number/FX523G5YIAUG4XN46FEH2KPU44/events.json","paper":"https://pith.science/paper/FX523G5Y"},"agent_actions":{"view_html":"https://pith.science/pith/FX523G5YIAUG4XN46FEH2KPU44","download_json":"https://pith.science/pith/FX523G5YIAUG4XN46FEH2KPU44.json","view_paper":"https://pith.science/paper/FX523G5Y","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.04079&json=true","fetch_graph":"https://pith.science/api/pith-number/FX523G5YIAUG4XN46FEH2KPU44/graph.json","fetch_events":"https://pith.science/api/pith-number/FX523G5YIAUG4XN46FEH2KPU44/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FX523G5YIAUG4XN46FEH2KPU44/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FX523G5YIAUG4XN46FEH2KPU44/action/storage_attestation","attest_author":"https://pith.science/pith/FX523G5YIAUG4XN46FEH2KPU44/action/author_attestation","sign_citation":"https://pith.science/pith/FX523G5YIAUG4XN46FEH2KPU44/action/citation_signature","submit_replication":"https://pith.science/pith/FX523G5YIAUG4XN46FEH2KPU44/action/replication_record"}},"created_at":"2026-07-05T11:22:41.468916+00:00","updated_at":"2026-07-05T11:22:41.468916+00:00"}