{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:5LRZSGJOVZECWO57SQPM67PXSI","short_pith_number":"pith:5LRZSGJO","schema_version":"1.0","canonical_sha256":"eae399192eae482b3bbf941ecf7df79224d747a5b63a5fee25ef9a5369fa813d","source":{"kind":"arxiv","id":"2011.01513","version":1},"attestation_state":"computed","paper":{"title":"CharBERT: Character-aware Pre-trained Language Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chenglei Si, Guoping Hu, Shijin Wang, Ting Liu, Wentao Ma, Yiming Cui","submitted_at":"2020-11-03T07:13:06Z","abstract_excerpt":"Most pre-trained language models (PLMs) construct word representations at subword level with Byte-Pair Encoding (BPE) or its variations, by which OOV (out-of-vocab) words are almost avoidable. However, those methods split a word into subword units and make the representation incomplete and fragile. In this paper, we propose a character-aware pre-trained language model named CharBERT improving on the previous methods (such as BERT, RoBERTa) to tackle these problems. We first construct the contextual word embedding for each token from the sequential character representations, then fuse the repre"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2011.01513","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2020-11-03T07:13:06Z","cross_cats_sorted":[],"title_canon_sha256":"37b7b7d3c112bf61f257a4afb90bc7f101af2b0b5cb3f7f645d753f1931ff41d","abstract_canon_sha256":"15f200eca56e3c902a19716479ccae36d87c7430f515a5eba64a8c96f17875ba"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:40:09.414866Z","signature_b64":"0MD3Ze6MrzSmz1mt9+P12VIdGmAPLEylOiYxSR6CRTyq2mc393W17CkjHOhYmdPjbZARAkLCByZ/R9ECg9LXBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"eae399192eae482b3bbf941ecf7df79224d747a5b63a5fee25ef9a5369fa813d","last_reissued_at":"2026-07-05T02:40:09.414473Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:40:09.414473Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CharBERT: Character-aware Pre-trained Language Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chenglei Si, Guoping Hu, Shijin Wang, Ting Liu, Wentao Ma, Yiming Cui","submitted_at":"2020-11-03T07:13:06Z","abstract_excerpt":"Most pre-trained language models (PLMs) construct word representations at subword level with Byte-Pair Encoding (BPE) or its variations, by which OOV (out-of-vocab) words are almost avoidable. However, those methods split a word into subword units and make the representation incomplete and fragile. In this paper, we propose a character-aware pre-trained language model named CharBERT improving on the previous methods (such as BERT, RoBERTa) to tackle these problems. We first construct the contextual word embedding for each token from the sequential character representations, then fuse the repre"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2011.01513","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2011.01513/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2011.01513","created_at":"2026-07-05T02:40:09.414530+00:00"},{"alias_kind":"arxiv_version","alias_value":"2011.01513v1","created_at":"2026-07-05T02:40:09.414530+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2011.01513","created_at":"2026-07-05T02:40:09.414530+00:00"},{"alias_kind":"pith_short_12","alias_value":"5LRZSGJOVZEC","created_at":"2026-07-05T02:40:09.414530+00:00"},{"alias_kind":"pith_short_16","alias_value":"5LRZSGJOVZECWO57","created_at":"2026-07-05T02:40:09.414530+00:00"},{"alias_kind":"pith_short_8","alias_value":"5LRZSGJO","created_at":"2026-07-05T02:40:09.414530+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.05468","citing_title":"TASE: Token Awareness and Structured Evaluation for Multilingual Language Models","ref_index":29,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5LRZSGJOVZECWO57SQPM67PXSI","json":"https://pith.science/pith/5LRZSGJOVZECWO57SQPM67PXSI.json","graph_json":"https://pith.science/api/pith-number/5LRZSGJOVZECWO57SQPM67PXSI/graph.json","events_json":"https://pith.science/api/pith-number/5LRZSGJOVZECWO57SQPM67PXSI/events.json","paper":"https://pith.science/paper/5LRZSGJO"},"agent_actions":{"view_html":"https://pith.science/pith/5LRZSGJOVZECWO57SQPM67PXSI","download_json":"https://pith.science/pith/5LRZSGJOVZECWO57SQPM67PXSI.json","view_paper":"https://pith.science/paper/5LRZSGJO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2011.01513&json=true","fetch_graph":"https://pith.science/api/pith-number/5LRZSGJOVZECWO57SQPM67PXSI/graph.json","fetch_events":"https://pith.science/api/pith-number/5LRZSGJOVZECWO57SQPM67PXSI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5LRZSGJOVZECWO57SQPM67PXSI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5LRZSGJOVZECWO57SQPM67PXSI/action/storage_attestation","attest_author":"https://pith.science/pith/5LRZSGJOVZECWO57SQPM67PXSI/action/author_attestation","sign_citation":"https://pith.science/pith/5LRZSGJOVZECWO57SQPM67PXSI/action/citation_signature","submit_replication":"https://pith.science/pith/5LRZSGJOVZECWO57SQPM67PXSI/action/replication_record"}},"created_at":"2026-07-05T02:40:09.414530+00:00","updated_at":"2026-07-05T02:40:09.414530+00:00"}