{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:JDRB2MUMDXHPXALUJ5EPGKW35A","short_pith_number":"pith:JDRB2MUM","schema_version":"1.0","canonical_sha256":"48e21d328c1dcefb81744f48f32adbe828f86a88f251736db1111914a4ce5d02","source":{"kind":"arxiv","id":"2605.20786","version":1},"attestation_state":"computed","paper":{"title":"Building Arabic NLP from the Ground Up: Twenty Years of Lessons, Failures, and Open Problems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Wajdi Zaghouani","submitted_at":"2026-05-20T06:30:16Z","abstract_excerpt":"This paper reflects on twenty years of building NLP resources and research infrastructure for Arabic, a language spoken by hundreds of millions yet historically underserved relative to languages such as English or Chinese. The first decade focused on foundational linguistic infrastructure; the second shifted toward computational social science, social media analysis, and socially oriented applications. Rather than cataloguing outputs, the paper examines what the experience of building them revealed. Three counterintuitive lessons emerge: building datasets is as much a social process as a techn"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2605.20786","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2026-05-20T06:30:16Z","cross_cats_sorted":[],"title_canon_sha256":"c92fc95e817ea34dc7a7654a80bb4c484ec3a93f5e9f055331ff8b86ca2e5cc7","abstract_canon_sha256":"ab567b8c74115e04a1accd27850d70e71e32f6bf7c389ea560ce39edbd6a29ef"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-21T01:04:54.211025Z","signature_b64":"VP8Rjv/IZtPYq3To6IiI6bUMzf83Bsaasme9LxDHK6lrJnFgli9kTL4suIsBkGayNKDZ4K2cwzCebZiljqnbCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"48e21d328c1dcefb81744f48f32adbe828f86a88f251736db1111914a4ce5d02","last_reissued_at":"2026-05-21T01:04:54.210178Z","signature_status":"signed_v1","first_computed_at":"2026-05-21T01:04:54.210178Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Building Arabic NLP from the Ground Up: Twenty Years of Lessons, Failures, and Open Problems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Wajdi Zaghouani","submitted_at":"2026-05-20T06:30:16Z","abstract_excerpt":"This paper reflects on twenty years of building NLP resources and research infrastructure for Arabic, a language spoken by hundreds of millions yet historically underserved relative to languages such as English or Chinese. The first decade focused on foundational linguistic infrastructure; the second shifted toward computational social science, social media analysis, and socially oriented applications. Rather than cataloguing outputs, the paper examines what the experience of building them revealed. Three counterintuitive lessons emerge: building datasets is as much a social process as a techn"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2605.20786","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2605.20786/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2605.20786","created_at":"2026-05-21T01:04:54.210325+00:00"},{"alias_kind":"arxiv_version","alias_value":"2605.20786v1","created_at":"2026-05-21T01:04:54.210325+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2605.20786","created_at":"2026-05-21T01:04:54.210325+00:00"},{"alias_kind":"pith_short_12","alias_value":"JDRB2MUMDXHP","created_at":"2026-05-21T01:04:54.210325+00:00"},{"alias_kind":"pith_short_16","alias_value":"JDRB2MUMDXHPXALU","created_at":"2026-05-21T01:04:54.210325+00:00"},{"alias_kind":"pith_short_8","alias_value":"JDRB2MUM","created_at":"2026-05-21T01:04:54.210325+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JDRB2MUMDXHPXALUJ5EPGKW35A","json":"https://pith.science/pith/JDRB2MUMDXHPXALUJ5EPGKW35A.json","graph_json":"https://pith.science/api/pith-number/JDRB2MUMDXHPXALUJ5EPGKW35A/graph.json","events_json":"https://pith.science/api/pith-number/JDRB2MUMDXHPXALUJ5EPGKW35A/events.json","paper":"https://pith.science/paper/JDRB2MUM"},"agent_actions":{"view_html":"https://pith.science/pith/JDRB2MUMDXHPXALUJ5EPGKW35A","download_json":"https://pith.science/pith/JDRB2MUMDXHPXALUJ5EPGKW35A.json","view_paper":"https://pith.science/paper/JDRB2MUM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2605.20786&json=true","fetch_graph":"https://pith.science/api/pith-number/JDRB2MUMDXHPXALUJ5EPGKW35A/graph.json","fetch_events":"https://pith.science/api/pith-number/JDRB2MUMDXHPXALUJ5EPGKW35A/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JDRB2MUMDXHPXALUJ5EPGKW35A/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JDRB2MUMDXHPXALUJ5EPGKW35A/action/storage_attestation","attest_author":"https://pith.science/pith/JDRB2MUMDXHPXALUJ5EPGKW35A/action/author_attestation","sign_citation":"https://pith.science/pith/JDRB2MUMDXHPXALUJ5EPGKW35A/action/citation_signature","submit_replication":"https://pith.science/pith/JDRB2MUMDXHPXALUJ5EPGKW35A/action/replication_record"}},"created_at":"2026-05-21T01:04:54.210325+00:00","updated_at":"2026-05-21T01:04:54.210325+00:00"}