{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:QJ2HIYGCKKBS3QXZKNJAACRP2J","short_pith_number":"pith:QJ2HIYGC","schema_version":"1.0","canonical_sha256":"82747460c252832dc2f95352000a2fd276a0f84969fbdd21d5b07bcd574fe8e8","source":{"kind":"arxiv","id":"2002.01207","version":1},"attestation_state":"computed","paper":{"title":"Arabic Diacritic Recovery Using a Feature-Rich biLSTM Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Ahmed Abdelali, Hamdy Mubarak, Kareem Darwish, Mohamed Eldesouki","submitted_at":"2020-02-04T10:09:42Z","abstract_excerpt":"Diacritics (short vowels) are typically omitted when writing Arabic text, and readers have to reintroduce them to correctly pronounce words. There are two types of Arabic diacritics: the first are core-word diacritics (CW), which specify the lexical selection, and the second are case endings (CE), which typically appear at the end of the word stem and generally specify their syntactic roles. Recovering CEs is relatively harder than recovering core-word diacritics due to inter-word dependencies, which are often distant. In this paper, we use a feature-rich recurrent neural network model that us"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2002.01207","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2020-02-04T10:09:42Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"d4b53553b31c50cb651b4c8ac68be62e77b5a1b54d89f7a5ad8502797e621f93","abstract_canon_sha256":"902b7b0298010de3eebae3e0cb996830977462bbf4d490e20af89e7aa58bb09c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:38:15.865640Z","signature_b64":"72Oo+H5GyZNqZs/ElmvBHHRDBnz90Uw12hDqWJDzID1j9xaw6SHr/N2rft1qsVeIG60uqCsNX34m3mH0jl8RBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"82747460c252832dc2f95352000a2fd276a0f84969fbdd21d5b07bcd574fe8e8","last_reissued_at":"2026-07-05T00:38:15.865054Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:38:15.865054Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Arabic Diacritic Recovery Using a Feature-Rich biLSTM Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Ahmed Abdelali, Hamdy Mubarak, Kareem Darwish, Mohamed Eldesouki","submitted_at":"2020-02-04T10:09:42Z","abstract_excerpt":"Diacritics (short vowels) are typically omitted when writing Arabic text, and readers have to reintroduce them to correctly pronounce words. There are two types of Arabic diacritics: the first are core-word diacritics (CW), which specify the lexical selection, and the second are case endings (CE), which typically appear at the end of the word stem and generally specify their syntactic roles. Recovering CEs is relatively harder than recovering core-word diacritics due to inter-word dependencies, which are often distant. In this paper, we use a feature-rich recurrent neural network model that us"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2002.01207","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2002.01207/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2002.01207","created_at":"2026-07-05T00:38:15.865130+00:00"},{"alias_kind":"arxiv_version","alias_value":"2002.01207v1","created_at":"2026-07-05T00:38:15.865130+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2002.01207","created_at":"2026-07-05T00:38:15.865130+00:00"},{"alias_kind":"pith_short_12","alias_value":"QJ2HIYGCKKBS","created_at":"2026-07-05T00:38:15.865130+00:00"},{"alias_kind":"pith_short_16","alias_value":"QJ2HIYGCKKBS3QXZ","created_at":"2026-07-05T00:38:15.865130+00:00"},{"alias_kind":"pith_short_8","alias_value":"QJ2HIYGC","created_at":"2026-07-05T00:38:15.865130+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2504.21635","citing_title":"Sadeed: Advancing Arabic Diacritization Through Small Language Model","ref_index":8,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QJ2HIYGCKKBS3QXZKNJAACRP2J","json":"https://pith.science/pith/QJ2HIYGCKKBS3QXZKNJAACRP2J.json","graph_json":"https://pith.science/api/pith-number/QJ2HIYGCKKBS3QXZKNJAACRP2J/graph.json","events_json":"https://pith.science/api/pith-number/QJ2HIYGCKKBS3QXZKNJAACRP2J/events.json","paper":"https://pith.science/paper/QJ2HIYGC"},"agent_actions":{"view_html":"https://pith.science/pith/QJ2HIYGCKKBS3QXZKNJAACRP2J","download_json":"https://pith.science/pith/QJ2HIYGCKKBS3QXZKNJAACRP2J.json","view_paper":"https://pith.science/paper/QJ2HIYGC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2002.01207&json=true","fetch_graph":"https://pith.science/api/pith-number/QJ2HIYGCKKBS3QXZKNJAACRP2J/graph.json","fetch_events":"https://pith.science/api/pith-number/QJ2HIYGCKKBS3QXZKNJAACRP2J/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QJ2HIYGCKKBS3QXZKNJAACRP2J/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QJ2HIYGCKKBS3QXZKNJAACRP2J/action/storage_attestation","attest_author":"https://pith.science/pith/QJ2HIYGCKKBS3QXZKNJAACRP2J/action/author_attestation","sign_citation":"https://pith.science/pith/QJ2HIYGCKKBS3QXZKNJAACRP2J/action/citation_signature","submit_replication":"https://pith.science/pith/QJ2HIYGCKKBS3QXZKNJAACRP2J/action/replication_record"}},"created_at":"2026-07-05T00:38:15.865130+00:00","updated_at":"2026-07-05T00:38:15.865130+00:00"}