{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:7VYQHLIO3HJURIXUQDL7QEM4OA","short_pith_number":"pith:7VYQHLIO","schema_version":"1.0","canonical_sha256":"fd7103ad0ed9d348a2f480d7f8119c700a5e10902777a4f6ecfa956358c5c2be","source":{"kind":"arxiv","id":"2012.15516","version":2},"attestation_state":"computed","paper":{"title":"AraELECTRA: Pre-Training Text Discriminators for Arabic Language Understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Fady Baly, Hazem Hajj, Wissam Antoun","submitted_at":"2020-12-31T09:35:39Z","abstract_excerpt":"Advances in English language representation enabled a more sample-efficient pre-training task by Efficiently Learning an Encoder that Classifies Token Replacements Accurately (ELECTRA). Which, instead of training a model to recover masked tokens, it trains a discriminator model to distinguish true input tokens from corrupted tokens that were replaced by a generator network. On the other hand, current Arabic language representation approaches rely only on pretraining via masked language modeling. In this paper, we develop an Arabic language representation model, which we name AraELECTRA. Our mo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2012.15516","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2020-12-31T09:35:39Z","cross_cats_sorted":[],"title_canon_sha256":"07eb60a4c21ce21edb2277bddecba607682d6087347c6352a8434996c76c8787","abstract_canon_sha256":"98999f64ffdc62b478f4f0916f4c562a4991f1187a1361b7b557fb019c561e94"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:21:00.216860Z","signature_b64":"5F8qr/3Q6AfatqsSSoj0IE5bQH8+Flt8UnJyGM43Ewnadbnhx4FFalAgan4+Eh51bfk74WxO10ovBzAQBJhvBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fd7103ad0ed9d348a2f480d7f8119c700a5e10902777a4f6ecfa956358c5c2be","last_reissued_at":"2026-07-05T02:21:00.216451Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:21:00.216451Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AraELECTRA: Pre-Training Text Discriminators for Arabic Language Understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Fady Baly, Hazem Hajj, Wissam Antoun","submitted_at":"2020-12-31T09:35:39Z","abstract_excerpt":"Advances in English language representation enabled a more sample-efficient pre-training task by Efficiently Learning an Encoder that Classifies Token Replacements Accurately (ELECTRA). Which, instead of training a model to recover masked tokens, it trains a discriminator model to distinguish true input tokens from corrupted tokens that were replaced by a generator network. On the other hand, current Arabic language representation approaches rely only on pretraining via masked language modeling. In this paper, we develop an Arabic language representation model, which we name AraELECTRA. Our mo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2012.15516","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2012.15516/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2012.15516","created_at":"2026-07-05T02:21:00.216502+00:00"},{"alias_kind":"arxiv_version","alias_value":"2012.15516v2","created_at":"2026-07-05T02:21:00.216502+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2012.15516","created_at":"2026-07-05T02:21:00.216502+00:00"},{"alias_kind":"pith_short_12","alias_value":"7VYQHLIO3HJU","created_at":"2026-07-05T02:21:00.216502+00:00"},{"alias_kind":"pith_short_16","alias_value":"7VYQHLIO3HJURIXU","created_at":"2026-07-05T02:21:00.216502+00:00"},{"alias_kind":"pith_short_8","alias_value":"7VYQHLIO","created_at":"2026-07-05T02:21:00.216502+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.23404","citing_title":"Enhanced Arabic Text Retrieval with Attentive Relevance Scoring","ref_index":22,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7VYQHLIO3HJURIXUQDL7QEM4OA","json":"https://pith.science/pith/7VYQHLIO3HJURIXUQDL7QEM4OA.json","graph_json":"https://pith.science/api/pith-number/7VYQHLIO3HJURIXUQDL7QEM4OA/graph.json","events_json":"https://pith.science/api/pith-number/7VYQHLIO3HJURIXUQDL7QEM4OA/events.json","paper":"https://pith.science/paper/7VYQHLIO"},"agent_actions":{"view_html":"https://pith.science/pith/7VYQHLIO3HJURIXUQDL7QEM4OA","download_json":"https://pith.science/pith/7VYQHLIO3HJURIXUQDL7QEM4OA.json","view_paper":"https://pith.science/paper/7VYQHLIO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2012.15516&json=true","fetch_graph":"https://pith.science/api/pith-number/7VYQHLIO3HJURIXUQDL7QEM4OA/graph.json","fetch_events":"https://pith.science/api/pith-number/7VYQHLIO3HJURIXUQDL7QEM4OA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7VYQHLIO3HJURIXUQDL7QEM4OA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7VYQHLIO3HJURIXUQDL7QEM4OA/action/storage_attestation","attest_author":"https://pith.science/pith/7VYQHLIO3HJURIXUQDL7QEM4OA/action/author_attestation","sign_citation":"https://pith.science/pith/7VYQHLIO3HJURIXUQDL7QEM4OA/action/citation_signature","submit_replication":"https://pith.science/pith/7VYQHLIO3HJURIXUQDL7QEM4OA/action/replication_record"}},"created_at":"2026-07-05T02:21:00.216502+00:00","updated_at":"2026-07-05T02:21:00.216502+00:00"}