{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:JDRB2MUMDXHPXALUJ5EPGKW35A","short_pith_number":"pith:JDRB2MUM","canonical_record":{"source":{"id":"2605.20786","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2026-05-20T06:30:16Z","cross_cats_sorted":[],"title_canon_sha256":"c92fc95e817ea34dc7a7654a80bb4c484ec3a93f5e9f055331ff8b86ca2e5cc7","abstract_canon_sha256":"ab567b8c74115e04a1accd27850d70e71e32f6bf7c389ea560ce39edbd6a29ef"},"schema_version":"1.0"},"canonical_sha256":"48e21d328c1dcefb81744f48f32adbe828f86a88f251736db1111914a4ce5d02","source":{"kind":"arxiv","id":"2605.20786","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2605.20786","created_at":"2026-05-21T01:04:54Z"},{"alias_kind":"arxiv_version","alias_value":"2605.20786v1","created_at":"2026-05-21T01:04:54Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2605.20786","created_at":"2026-05-21T01:04:54Z"},{"alias_kind":"pith_short_12","alias_value":"JDRB2MUMDXHP","created_at":"2026-05-21T01:04:54Z"},{"alias_kind":"pith_short_16","alias_value":"JDRB2MUMDXHPXALU","created_at":"2026-05-21T01:04:54Z"},{"alias_kind":"pith_short_8","alias_value":"JDRB2MUM","created_at":"2026-05-21T01:04:54Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:JDRB2MUMDXHPXALUJ5EPGKW35A","target":"record","payload":{"canonical_record":{"source":{"id":"2605.20786","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2026-05-20T06:30:16Z","cross_cats_sorted":[],"title_canon_sha256":"c92fc95e817ea34dc7a7654a80bb4c484ec3a93f5e9f055331ff8b86ca2e5cc7","abstract_canon_sha256":"ab567b8c74115e04a1accd27850d70e71e32f6bf7c389ea560ce39edbd6a29ef"},"schema_version":"1.0"},"canonical_sha256":"48e21d328c1dcefb81744f48f32adbe828f86a88f251736db1111914a4ce5d02","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-21T01:04:54.211025Z","signature_b64":"VP8Rjv/IZtPYq3To6IiI6bUMzf83Bsaasme9LxDHK6lrJnFgli9kTL4suIsBkGayNKDZ4K2cwzCebZiljqnbCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"48e21d328c1dcefb81744f48f32adbe828f86a88f251736db1111914a4ce5d02","last_reissued_at":"2026-05-21T01:04:54.210178Z","signature_status":"signed_v1","first_computed_at":"2026-05-21T01:04:54.210178Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2605.20786","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-21T01:04:54Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"RDNrrVOsWQMydrm405/ABas2nXQBEnUHIf9F2Xg8ppczTwgUCq36hEwHBYkJN16lxUJ3n0Qiy1zsB/3tMMSABg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-05-24T06:10:47.903480Z"},"content_sha256":"4ae16f12ddf75936453fcc73166ac5992aae44f3ef360eb0d7a52cdf50f9d19e","schema_version":"1.0","event_id":"sha256:4ae16f12ddf75936453fcc73166ac5992aae44f3ef360eb0d7a52cdf50f9d19e"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:JDRB2MUMDXHPXALUJ5EPGKW35A","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Building Arabic NLP from the Ground Up: Twenty Years of Lessons, Failures, and Open Problems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Wajdi Zaghouani","submitted_at":"2026-05-20T06:30:16Z","abstract_excerpt":"This paper reflects on twenty years of building NLP resources and research infrastructure for Arabic, a language spoken by hundreds of millions yet historically underserved relative to languages such as English or Chinese. The first decade focused on foundational linguistic infrastructure; the second shifted toward computational social science, social media analysis, and socially oriented applications. Rather than cataloguing outputs, the paper examines what the experience of building them revealed. Three counterintuitive lessons emerge: building datasets is as much a social process as a techn"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2605.20786","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2605.20786/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-21T01:04:54Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"igjxwgxzeBSgH1hlLSF4R4MoT7TkYSiTRauYDTSo5HOMfTeabARi0GyTzeggmZu3BCQlF+dETheLQX5enL8NAw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-05-24T06:10:47.904102Z"},"content_sha256":"89a5cf4b0dbf1e54e00f9eb024a082cc77e60e3d81a33a7831d29318fd002516","schema_version":"1.0","event_id":"sha256:89a5cf4b0dbf1e54e00f9eb024a082cc77e60e3d81a33a7831d29318fd002516"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/JDRB2MUMDXHPXALUJ5EPGKW35A/bundle.json","state_url":"https://pith.science/pith/JDRB2MUMDXHPXALUJ5EPGKW35A/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/JDRB2MUMDXHPXALUJ5EPGKW35A/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-05-24T06:10:47Z","links":{"resolver":"https://pith.science/pith/JDRB2MUMDXHPXALUJ5EPGKW35A","bundle":"https://pith.science/pith/JDRB2MUMDXHPXALUJ5EPGKW35A/bundle.json","state":"https://pith.science/pith/JDRB2MUMDXHPXALUJ5EPGKW35A/state.json","well_known_bundle":"https://pith.science/.well-known/pith/JDRB2MUMDXHPXALUJ5EPGKW35A/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:JDRB2MUMDXHPXALUJ5EPGKW35A","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"ab567b8c74115e04a1accd27850d70e71e32f6bf7c389ea560ce39edbd6a29ef","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2026-05-20T06:30:16Z","title_canon_sha256":"c92fc95e817ea34dc7a7654a80bb4c484ec3a93f5e9f055331ff8b86ca2e5cc7"},"schema_version":"1.0","source":{"id":"2605.20786","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2605.20786","created_at":"2026-05-21T01:04:54Z"},{"alias_kind":"arxiv_version","alias_value":"2605.20786v1","created_at":"2026-05-21T01:04:54Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2605.20786","created_at":"2026-05-21T01:04:54Z"},{"alias_kind":"pith_short_12","alias_value":"JDRB2MUMDXHP","created_at":"2026-05-21T01:04:54Z"},{"alias_kind":"pith_short_16","alias_value":"JDRB2MUMDXHPXALU","created_at":"2026-05-21T01:04:54Z"},{"alias_kind":"pith_short_8","alias_value":"JDRB2MUM","created_at":"2026-05-21T01:04:54Z"}],"graph_snapshots":[{"event_id":"sha256:89a5cf4b0dbf1e54e00f9eb024a082cc77e60e3d81a33a7831d29318fd002516","target":"graph","created_at":"2026-05-21T01:04:54Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2605.20786/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"This paper reflects on twenty years of building NLP resources and research infrastructure for Arabic, a language spoken by hundreds of millions yet historically underserved relative to languages such as English or Chinese. The first decade focused on foundational linguistic infrastructure; the second shifted toward computational social science, social media analysis, and socially oriented applications. Rather than cataloguing outputs, the paper examines what the experience of building them revealed. Three counterintuitive lessons emerge: building datasets is as much a social process as a techn","authors_text":"Wajdi Zaghouani","cross_cats":[],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2026-05-20T06:30:16Z","title":"Building Arabic NLP from the Ground Up: Twenty Years of Lessons, Failures, and Open Problems"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2605.20786","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:4ae16f12ddf75936453fcc73166ac5992aae44f3ef360eb0d7a52cdf50f9d19e","target":"record","created_at":"2026-05-21T01:04:54Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"ab567b8c74115e04a1accd27850d70e71e32f6bf7c389ea560ce39edbd6a29ef","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2026-05-20T06:30:16Z","title_canon_sha256":"c92fc95e817ea34dc7a7654a80bb4c484ec3a93f5e9f055331ff8b86ca2e5cc7"},"schema_version":"1.0","source":{"id":"2605.20786","kind":"arxiv","version":1}},"canonical_sha256":"48e21d328c1dcefb81744f48f32adbe828f86a88f251736db1111914a4ce5d02","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"48e21d328c1dcefb81744f48f32adbe828f86a88f251736db1111914a4ce5d02","first_computed_at":"2026-05-21T01:04:54.210178Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-05-21T01:04:54.210178Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"VP8Rjv/IZtPYq3To6IiI6bUMzf83Bsaasme9LxDHK6lrJnFgli9kTL4suIsBkGayNKDZ4K2cwzCebZiljqnbCw==","signature_status":"signed_v1","signed_at":"2026-05-21T01:04:54.211025Z","signed_message":"canonical_sha256_bytes"},"source_id":"2605.20786","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:4ae16f12ddf75936453fcc73166ac5992aae44f3ef360eb0d7a52cdf50f9d19e","sha256:89a5cf4b0dbf1e54e00f9eb024a082cc77e60e3d81a33a7831d29318fd002516"],"state_sha256":"f0a05cf63fff51565f90941965473b40e58c27d47fb2af180509917e69f66d5a"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"sLSNTfyRzveoX87aVw54/8yqhuIYMSI/sMc17GgVjBod+zDXryr6FrQ323JQ+7Q2N7AmUvYU2PBLBu5CGDquDg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-05-24T06:10:47.907505Z","bundle_sha256":"968970ed570578d32dc98f7e07163e814809a562b977bd69370dae697e4309ea"}}