{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:PK3E3VS4BCJP3QZZ3BBS63VM47","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"2d1498f17f5a6bffb61872bc70f31821003c866c7fe840a192866cd4fba803f9","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-03-07T15:13:58Z","title_canon_sha256":"62075248182ff341edad392de809b38c0d0bfdd3b81d03794f3c3c837427cccc"},"schema_version":"1.0","source":{"id":"2503.05500","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2503.05500","created_at":"2026-06-02T02:04:46Z"},{"alias_kind":"arxiv_version","alias_value":"2503.05500v3","created_at":"2026-06-02T02:04:46Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.05500","created_at":"2026-06-02T02:04:46Z"},{"alias_kind":"pith_short_12","alias_value":"PK3E3VS4BCJP","created_at":"2026-06-02T02:04:46Z"},{"alias_kind":"pith_short_16","alias_value":"PK3E3VS4BCJP3QZZ","created_at":"2026-06-02T02:04:46Z"},{"alias_kind":"pith_short_8","alias_value":"PK3E3VS4","created_at":"2026-06-02T02:04:46Z"}],"graph_snapshots":[{"event_id":"sha256:f9a5465f2fcffb388bab863ee917d5d68610192ea862469fd3d4eba7267f828f","target":"graph","created_at":"2026-06-02T02:04:46Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2503.05500/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"General-purpose multilingual vector representations, used in retrieval, regression and classification, are traditionally obtained from bidirectional encoder models. Despite their wide applicability, encoders have been recently overshadowed by advances in generative decoder-only models. However, many innovations driving this progress are not inherently tied to decoders. In this paper, we revisit the development of multilingual encoders through the lens of these advances, and introduce EuroBERT, a family of multilingual encoders covering European and widely spoken global languages. Our models ou","authors_text":"Andr\\'e Martins, Ayoub Hammal, Caio Corro, C\\'eline Hudelot, Duarte M. Alves, Emmanuel Malherbe, Etienne Malaboeuf, Fanny Jourdan, Gabriel Hautreux, Hippolyte Gisserot-Boukhlef, Jo\\~ao Alves, Kevin El Haddad, Manuel Faysse, Maxime Peyrard, Nicolas Boizard, Nuno M. Guerreiro, Patrick Fernandes, Pierre Colombo, Ricardo Rei","cross_cats":["cs.AI"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-03-07T15:13:58Z","title":"EuroBERT: Scaling Multilingual Encoders for European Languages"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.05500","kind":"arxiv","version":3},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:18ba8f8f40f6c612477caa3e6a3bed1750b5f5584e156de1a4ea97858c7c6c79","target":"record","created_at":"2026-06-02T02:04:46Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"2d1498f17f5a6bffb61872bc70f31821003c866c7fe840a192866cd4fba803f9","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-03-07T15:13:58Z","title_canon_sha256":"62075248182ff341edad392de809b38c0d0bfdd3b81d03794f3c3c837427cccc"},"schema_version":"1.0","source":{"id":"2503.05500","kind":"arxiv","version":3}},"canonical_sha256":"7ab64dd65c0892fdc339d8432f6eace7e166cf3051951df67dfcc36ca1dff000","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"7ab64dd65c0892fdc339d8432f6eace7e166cf3051951df67dfcc36ca1dff000","first_computed_at":"2026-06-02T02:04:46.572717Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-06-02T02:04:46.572717Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"rRJaacHTm1541/8ZyNzWGEpyvwdlmOV3Br3CIUeMur41Rs9sbZUQnK2SS1fPMRY7DT2I1B0EZNGhsrDV2Si8Ag==","signature_status":"signed_v1","signed_at":"2026-06-02T02:04:46.573255Z","signed_message":"canonical_sha256_bytes"},"source_id":"2503.05500","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:18ba8f8f40f6c612477caa3e6a3bed1750b5f5584e156de1a4ea97858c7c6c79","sha256:f9a5465f2fcffb388bab863ee917d5d68610192ea862469fd3d4eba7267f828f"],"state_sha256":"219b2a4babf4abbd800920462a2bf0a263556c3faf4e94abf1b41d6f96b0acef"}