{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:X47RQZ3YL2QBQXE6BO3IV5LIDP","short_pith_number":"pith:X47RQZ3Y","schema_version":"1.0","canonical_sha256":"bf3f1867785ea0185c9e0bb68af5681be8bb14cc1afdec4c68d4d2470e4d1188","source":{"kind":"arxiv","id":"2411.08868","version":1},"attestation_state":"computed","paper":{"title":"CamemBERT 2.0: A Smarter French Language Model Aged to Perfection","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Beno\\^it Sagot, Djam\\'e Seddah, \\'Eric de la Clergerie, Francis Kulumba, Rian Touchent, Wissam Antoun","submitted_at":"2024-11-13T18:49:35Z","abstract_excerpt":"French language models, such as CamemBERT, have been widely adopted across industries for natural language processing (NLP) tasks, with models like CamemBERT seeing over 4 million downloads per month. However, these models face challenges due to temporal concept drift, where outdated training data leads to a decline in performance, especially when encountering new topics and terminology. This issue emphasizes the need for updated models that reflect current linguistic trends. In this paper, we introduce two new versions of the CamemBERT base model-CamemBERTav2 and CamemBERTv2-designed to addre"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.08868","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-11-13T18:49:35Z","cross_cats_sorted":[],"title_canon_sha256":"dfa658d94c832bd986d0238ee15bfd4d6fd64829b826d1a46327ea50e964d0d6","abstract_canon_sha256":"87c134f85872d2a821e9bbcf3367362dbcf18464eabc6249d7f85a73ff52c9fc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:35:03.879754Z","signature_b64":"zUcFCvqhXwxdVPnAtfWDr9PVfXOo3Vj9wkChIQdbXzzvOfsiXuXpHRFfmxJ8NR6vJfgrgGtrh/2WmMDNyFybDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bf3f1867785ea0185c9e0bb68af5681be8bb14cc1afdec4c68d4d2470e4d1188","last_reissued_at":"2026-07-05T09:35:03.879310Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:35:03.879310Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CamemBERT 2.0: A Smarter French Language Model Aged to Perfection","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Beno\\^it Sagot, Djam\\'e Seddah, \\'Eric de la Clergerie, Francis Kulumba, Rian Touchent, Wissam Antoun","submitted_at":"2024-11-13T18:49:35Z","abstract_excerpt":"French language models, such as CamemBERT, have been widely adopted across industries for natural language processing (NLP) tasks, with models like CamemBERT seeing over 4 million downloads per month. However, these models face challenges due to temporal concept drift, where outdated training data leads to a decline in performance, especially when encountering new topics and terminology. This issue emphasizes the need for updated models that reflect current linguistic trends. In this paper, we introduce two new versions of the CamemBERT base model-CamemBERTav2 and CamemBERTv2-designed to addre"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.08868","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.08868/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.08868","created_at":"2026-07-05T09:35:03.879372+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.08868v1","created_at":"2026-07-05T09:35:03.879372+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.08868","created_at":"2026-07-05T09:35:03.879372+00:00"},{"alias_kind":"pith_short_12","alias_value":"X47RQZ3YL2QB","created_at":"2026-07-05T09:35:03.879372+00:00"},{"alias_kind":"pith_short_16","alias_value":"X47RQZ3YL2QBQXE6","created_at":"2026-07-05T09:35:03.879372+00:00"},{"alias_kind":"pith_short_8","alias_value":"X47RQZ3Y","created_at":"2026-07-05T09:35:03.879372+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22722","citing_title":"moBERTo: A Modern Encoder for Portuguese via Continued Pretraining of ModernBERT","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2407.20595","citing_title":"HALvest-Contrastive: Retrieval-Like Authorship Attribution with Patch-Level Late Interaction","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18226","citing_title":"Model in Distress: Sentiment Analysis on French Synthetic Social Media","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/X47RQZ3YL2QBQXE6BO3IV5LIDP","json":"https://pith.science/pith/X47RQZ3YL2QBQXE6BO3IV5LIDP.json","graph_json":"https://pith.science/api/pith-number/X47RQZ3YL2QBQXE6BO3IV5LIDP/graph.json","events_json":"https://pith.science/api/pith-number/X47RQZ3YL2QBQXE6BO3IV5LIDP/events.json","paper":"https://pith.science/paper/X47RQZ3Y"},"agent_actions":{"view_html":"https://pith.science/pith/X47RQZ3YL2QBQXE6BO3IV5LIDP","download_json":"https://pith.science/pith/X47RQZ3YL2QBQXE6BO3IV5LIDP.json","view_paper":"https://pith.science/paper/X47RQZ3Y","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.08868&json=true","fetch_graph":"https://pith.science/api/pith-number/X47RQZ3YL2QBQXE6BO3IV5LIDP/graph.json","fetch_events":"https://pith.science/api/pith-number/X47RQZ3YL2QBQXE6BO3IV5LIDP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/X47RQZ3YL2QBQXE6BO3IV5LIDP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/X47RQZ3YL2QBQXE6BO3IV5LIDP/action/storage_attestation","attest_author":"https://pith.science/pith/X47RQZ3YL2QBQXE6BO3IV5LIDP/action/author_attestation","sign_citation":"https://pith.science/pith/X47RQZ3YL2QBQXE6BO3IV5LIDP/action/citation_signature","submit_replication":"https://pith.science/pith/X47RQZ3YL2QBQXE6BO3IV5LIDP/action/replication_record"}},"created_at":"2026-07-05T09:35:03.879372+00:00","updated_at":"2026-07-05T09:35:03.879372+00:00"}