{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:E2LOO2N3TNPAG73AGWYB47KU43","short_pith_number":"pith:E2LOO2N3","schema_version":"1.0","canonical_sha256":"2696e769bb9b5e037f6035b01e7d54e6c619db13a72271b7003b26817d86b9a5","source":{"kind":"arxiv","id":"2406.05760","version":1},"attestation_state":"computed","paper":{"title":"Arabic Diacritics in the Wild: Exploiting Opportunities for Improved Diacritization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Go Inoue, Nizar Habash, Ossama Obeid, Salman Elgamal, Tameem Kabbani","submitted_at":"2024-06-09T12:29:55Z","abstract_excerpt":"The widespread absence of diacritical marks in Arabic text poses a significant challenge for Arabic natural language processing (NLP). This paper explores instances of naturally occurring diacritics, referred to as \"diacritics in the wild,\" to unveil patterns and latent information across six diverse genres: news articles, novels, children's books, poetry, political documents, and ChatGPT outputs. We present a new annotated dataset that maps real-world partially diacritized words to their maximal full diacritization in context. Additionally, we propose extensions to the analyze-and-disambiguat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.05760","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-06-09T12:29:55Z","cross_cats_sorted":[],"title_canon_sha256":"a8576d6e5ad93b8ad009aaffa79c06976397521b79b1749e8d107591b6e12eef","abstract_canon_sha256":"333724393ef5b0c7380a23c83e7966d00cb8837875753b4e180a41270fb72042"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:29:33.382271Z","signature_b64":"S01GU38yHsATyBgMdTL/hZ8MlIFRfxlK9aQXRUVH05JbwTDwJ4z16dURCkIrixQctU4Z0qjfo1k39IS/9q3OBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2696e769bb9b5e037f6035b01e7d54e6c619db13a72271b7003b26817d86b9a5","last_reissued_at":"2026-07-05T08:29:33.381893Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:29:33.381893Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Arabic Diacritics in the Wild: Exploiting Opportunities for Improved Diacritization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Go Inoue, Nizar Habash, Ossama Obeid, Salman Elgamal, Tameem Kabbani","submitted_at":"2024-06-09T12:29:55Z","abstract_excerpt":"The widespread absence of diacritical marks in Arabic text poses a significant challenge for Arabic natural language processing (NLP). This paper explores instances of naturally occurring diacritics, referred to as \"diacritics in the wild,\" to unveil patterns and latent information across six diverse genres: news articles, novels, children's books, poetry, political documents, and ChatGPT outputs. We present a new annotated dataset that maps real-world partially diacritized words to their maximal full diacritization in context. Additionally, we propose extensions to the analyze-and-disambiguat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.05760","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.05760/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.05760","created_at":"2026-07-05T08:29:33.381948+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.05760v1","created_at":"2026-07-05T08:29:33.381948+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.05760","created_at":"2026-07-05T08:29:33.381948+00:00"},{"alias_kind":"pith_short_12","alias_value":"E2LOO2N3TNPA","created_at":"2026-07-05T08:29:33.381948+00:00"},{"alias_kind":"pith_short_16","alias_value":"E2LOO2N3TNPAG73A","created_at":"2026-07-05T08:29:33.381948+00:00"},{"alias_kind":"pith_short_8","alias_value":"E2LOO2N3","created_at":"2026-07-05T08:29:33.381948+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.18399","citing_title":"Lemmatization as a Classification Task: Results from Arabic across Multiple Genres","ref_index":13,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/E2LOO2N3TNPAG73AGWYB47KU43","json":"https://pith.science/pith/E2LOO2N3TNPAG73AGWYB47KU43.json","graph_json":"https://pith.science/api/pith-number/E2LOO2N3TNPAG73AGWYB47KU43/graph.json","events_json":"https://pith.science/api/pith-number/E2LOO2N3TNPAG73AGWYB47KU43/events.json","paper":"https://pith.science/paper/E2LOO2N3"},"agent_actions":{"view_html":"https://pith.science/pith/E2LOO2N3TNPAG73AGWYB47KU43","download_json":"https://pith.science/pith/E2LOO2N3TNPAG73AGWYB47KU43.json","view_paper":"https://pith.science/paper/E2LOO2N3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.05760&json=true","fetch_graph":"https://pith.science/api/pith-number/E2LOO2N3TNPAG73AGWYB47KU43/graph.json","fetch_events":"https://pith.science/api/pith-number/E2LOO2N3TNPAG73AGWYB47KU43/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/E2LOO2N3TNPAG73AGWYB47KU43/action/timestamp_anchor","attest_storage":"https://pith.science/pith/E2LOO2N3TNPAG73AGWYB47KU43/action/storage_attestation","attest_author":"https://pith.science/pith/E2LOO2N3TNPAG73AGWYB47KU43/action/author_attestation","sign_citation":"https://pith.science/pith/E2LOO2N3TNPAG73AGWYB47KU43/action/citation_signature","submit_replication":"https://pith.science/pith/E2LOO2N3TNPAG73AGWYB47KU43/action/replication_record"}},"created_at":"2026-07-05T08:29:33.381948+00:00","updated_at":"2026-07-05T08:29:33.381948+00:00"}