{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:C6W5OZB6INIS25ARUQB63UDBQJ","short_pith_number":"pith:C6W5OZB6","schema_version":"1.0","canonical_sha256":"17add7643e43512d7411a403edd061826094cef43b16643b1abe9efcc6a8936d","source":{"kind":"arxiv","id":"2509.06888","version":1},"attestation_state":"computed","paper":{"title":"mmBERT: A Modern Multilingual Encoder with Annealed Language Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.IR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Benjamin Van Durme, Dawn Lawrie, Eugene Yang, Marc Marone, Orion Weller, William Fleshman","submitted_at":"2025-09-08T17:08:42Z","abstract_excerpt":"Encoder-only languages models are frequently used for a variety of standard machine learning tasks, including classification and retrieval. However, there has been a lack of recent research for encoder models, especially with respect to multilingual models. We introduce mmBERT, an encoder-only language model pretrained on 3T tokens of multilingual text in over 1800 languages. To build mmBERT we introduce several novel elements, including an inverse mask ratio schedule and an inverse temperature sampling ratio. We add over 1700 low-resource languages to the data mix only during the decay phase,"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2509.06888","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-09-08T17:08:42Z","cross_cats_sorted":["cs.IR","cs.LG"],"title_canon_sha256":"a916f0a8ab648a3b52198581266a20e1e794257d9d3cf5c112bd4c8866b14e6c","abstract_canon_sha256":"36f4516224f119a50674f7558450d5767cbdc983acabfd469e1b24cfe36e89de"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:06:54.577244Z","signature_b64":"B82guS+NNazAx8fWU2UaR0uYSQyFz2WAxp+xle7iyt/vTiZ6eI6qzDLsb8zeLNOtfWsdeo3PaD+QunhPYsP8DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"17add7643e43512d7411a403edd061826094cef43b16643b1abe9efcc6a8936d","last_reissued_at":"2026-07-05T12:06:54.576731Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:06:54.576731Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"mmBERT: A Modern Multilingual Encoder with Annealed Language Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.IR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Benjamin Van Durme, Dawn Lawrie, Eugene Yang, Marc Marone, Orion Weller, William Fleshman","submitted_at":"2025-09-08T17:08:42Z","abstract_excerpt":"Encoder-only languages models are frequently used for a variety of standard machine learning tasks, including classification and retrieval. However, there has been a lack of recent research for encoder models, especially with respect to multilingual models. We introduce mmBERT, an encoder-only language model pretrained on 3T tokens of multilingual text in over 1800 languages. To build mmBERT we introduce several novel elements, including an inverse mask ratio schedule and an inverse temperature sampling ratio. We add over 1700 low-resource languages to the data mix only during the decay phase,"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.06888","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.06888/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2509.06888","created_at":"2026-07-05T12:06:54.576789+00:00"},{"alias_kind":"arxiv_version","alias_value":"2509.06888v1","created_at":"2026-07-05T12:06:54.576789+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.06888","created_at":"2026-07-05T12:06:54.576789+00:00"},{"alias_kind":"pith_short_12","alias_value":"C6W5OZB6INIS","created_at":"2026-07-05T12:06:54.576789+00:00"},{"alias_kind":"pith_short_16","alias_value":"C6W5OZB6INIS25AR","created_at":"2026-07-05T12:06:54.576789+00:00"},{"alias_kind":"pith_short_8","alias_value":"C6W5OZB6","created_at":"2026-07-05T12:06:54.576789+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.06738","citing_title":"Modular Monolingual Adaptation using Pretrained Language Models","ref_index":93,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06349","citing_title":"\"Chi nas dal soch el sent de legn\" -- Auditing Text Corpora for Lombard","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30312","citing_title":"DialogPII: A multilingual dataset of synthetic dialog transcripts to detect personal information","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23033","citing_title":"Uncovering the Latent Potential of Deep Intermediate Representations","ref_index":86,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14257","citing_title":"Sakura at BEA 2026 Shared Task 1: What Makes Vocabulary Difficult?","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2510.27269","citing_title":"Why Do Multilingual Reasoning Gaps Emerge in Reasoning Language Models?","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14257","citing_title":"Sakura at BEA 2026 Shared Task 1: What Makes Vocabulary Difficult?","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16380","citing_title":"Data Mixing for Large Language Models Pretraining: A Survey and Outlook","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13521","citing_title":"Granite Embedding Multilingual R2 Models","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05277","citing_title":"GLiNER Guard: Unified Encoder Family for Production LLM Safety and Privacy","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/C6W5OZB6INIS25ARUQB63UDBQJ","json":"https://pith.science/pith/C6W5OZB6INIS25ARUQB63UDBQJ.json","graph_json":"https://pith.science/api/pith-number/C6W5OZB6INIS25ARUQB63UDBQJ/graph.json","events_json":"https://pith.science/api/pith-number/C6W5OZB6INIS25ARUQB63UDBQJ/events.json","paper":"https://pith.science/paper/C6W5OZB6"},"agent_actions":{"view_html":"https://pith.science/pith/C6W5OZB6INIS25ARUQB63UDBQJ","download_json":"https://pith.science/pith/C6W5OZB6INIS25ARUQB63UDBQJ.json","view_paper":"https://pith.science/paper/C6W5OZB6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2509.06888&json=true","fetch_graph":"https://pith.science/api/pith-number/C6W5OZB6INIS25ARUQB63UDBQJ/graph.json","fetch_events":"https://pith.science/api/pith-number/C6W5OZB6INIS25ARUQB63UDBQJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/C6W5OZB6INIS25ARUQB63UDBQJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/C6W5OZB6INIS25ARUQB63UDBQJ/action/storage_attestation","attest_author":"https://pith.science/pith/C6W5OZB6INIS25ARUQB63UDBQJ/action/author_attestation","sign_citation":"https://pith.science/pith/C6W5OZB6INIS25ARUQB63UDBQJ/action/citation_signature","submit_replication":"https://pith.science/pith/C6W5OZB6INIS25ARUQB63UDBQJ/action/replication_record"}},"created_at":"2026-07-05T12:06:54.576789+00:00","updated_at":"2026-07-05T12:06:54.576789+00:00"}