{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:TGGQFJJY2LI4JOJLIYVHVGCVT3","short_pith_number":"pith:TGGQFJJY","schema_version":"1.0","canonical_sha256":"998d02a538d2d1c4b92b462a7a98559edff222799d3d7d4006a23b883cc3eeb3","source":{"kind":"arxiv","id":"2001.06286","version":2},"attestation_state":"computed","paper":{"title":"RobBERT: a Dutch RoBERTa-based Language Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Bettina Berendt, Pieter Delobelle, Thomas Winters","submitted_at":"2020-01-17T13:25:44Z","abstract_excerpt":"Pre-trained language models have been dominating the field of natural language processing in recent years, and have led to significant performance gains for various complex natural language tasks. One of the most prominent pre-trained language models is BERT, which was released as an English as well as a multilingual version. Although multilingual BERT performs well on many tasks, recent studies show that BERT models trained on a single language significantly outperform the multilingual version. Training a Dutch BERT model thus has a lot of potential for a wide range of Dutch NLP tasks. While "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2001.06286","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2020-01-17T13:25:44Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"88c3b67134ef43aba78bfc29b50b0626579c76b4015947275c8aac2dc875704b","abstract_canon_sha256":"6a693cd4c0a29d1fd131b6fbdb928523d71aeedf797f382389728406db33372b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:35:44.662864Z","signature_b64":"GFh3vkDiWE/vjG9YnVWm0QWc10Dff7u8yknHW0jle7BArg9YBtlWKQQpEsY26VrEkUTrTd7Ap+Q3Vgu/3MFXCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"998d02a538d2d1c4b92b462a7a98559edff222799d3d7d4006a23b883cc3eeb3","last_reissued_at":"2026-07-05T01:35:44.662472Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:35:44.662472Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RobBERT: a Dutch RoBERTa-based Language Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Bettina Berendt, Pieter Delobelle, Thomas Winters","submitted_at":"2020-01-17T13:25:44Z","abstract_excerpt":"Pre-trained language models have been dominating the field of natural language processing in recent years, and have led to significant performance gains for various complex natural language tasks. One of the most prominent pre-trained language models is BERT, which was released as an English as well as a multilingual version. Although multilingual BERT performs well on many tasks, recent studies show that BERT models trained on a single language significantly outperform the multilingual version. Training a Dutch BERT model thus has a lot of potential for a wide range of Dutch NLP tasks. While "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2001.06286","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2001.06286/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2001.06286","created_at":"2026-07-05T01:35:44.662534+00:00"},{"alias_kind":"arxiv_version","alias_value":"2001.06286v2","created_at":"2026-07-05T01:35:44.662534+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2001.06286","created_at":"2026-07-05T01:35:44.662534+00:00"},{"alias_kind":"pith_short_12","alias_value":"TGGQFJJY2LI4","created_at":"2026-07-05T01:35:44.662534+00:00"},{"alias_kind":"pith_short_16","alias_value":"TGGQFJJY2LI4JOJL","created_at":"2026-07-05T01:35:44.662534+00:00"},{"alias_kind":"pith_short_8","alias_value":"TGGQFJJY","created_at":"2026-07-05T01:35:44.662534+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.02100","citing_title":"PortBERT: Navigating the Depths of Portuguese Language Models","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21555","citing_title":"Finding Meaning in Embeddings: Concept Separation Curves","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TGGQFJJY2LI4JOJLIYVHVGCVT3","json":"https://pith.science/pith/TGGQFJJY2LI4JOJLIYVHVGCVT3.json","graph_json":"https://pith.science/api/pith-number/TGGQFJJY2LI4JOJLIYVHVGCVT3/graph.json","events_json":"https://pith.science/api/pith-number/TGGQFJJY2LI4JOJLIYVHVGCVT3/events.json","paper":"https://pith.science/paper/TGGQFJJY"},"agent_actions":{"view_html":"https://pith.science/pith/TGGQFJJY2LI4JOJLIYVHVGCVT3","download_json":"https://pith.science/pith/TGGQFJJY2LI4JOJLIYVHVGCVT3.json","view_paper":"https://pith.science/paper/TGGQFJJY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2001.06286&json=true","fetch_graph":"https://pith.science/api/pith-number/TGGQFJJY2LI4JOJLIYVHVGCVT3/graph.json","fetch_events":"https://pith.science/api/pith-number/TGGQFJJY2LI4JOJLIYVHVGCVT3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TGGQFJJY2LI4JOJLIYVHVGCVT3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TGGQFJJY2LI4JOJLIYVHVGCVT3/action/storage_attestation","attest_author":"https://pith.science/pith/TGGQFJJY2LI4JOJLIYVHVGCVT3/action/author_attestation","sign_citation":"https://pith.science/pith/TGGQFJJY2LI4JOJLIYVHVGCVT3/action/citation_signature","submit_replication":"https://pith.science/pith/TGGQFJJY2LI4JOJLIYVHVGCVT3/action/replication_record"}},"created_at":"2026-07-05T01:35:44.662534+00:00","updated_at":"2026-07-05T01:35:44.662534+00:00"}