{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:2HOCQMB4UJY2RPD64PUUHLUOQP","short_pith_number":"pith:2HOCQMB4","schema_version":"1.0","canonical_sha256":"d1dc28303ca271a8bc7ee3e943ae8e83c451b846df51ef2e72de545773534da3","source":{"kind":"arxiv","id":"2507.15850","version":3},"attestation_state":"computed","paper":{"title":"3LM: Bridging Arabic, STEM, and Code through Benchmarking","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ahmed Alzubaidi, Basma El Amel Boussaha, Giulia Campesan, Hakim Hacid, Leen AlQadi, Mohammed Alyafeai, Mugariya Farooq, Shaikha Alsuwaidi","submitted_at":"2025-07-21T17:58:27Z","abstract_excerpt":"Arabic is one of the most widely spoken languages in the world, yet efforts to develop and evaluate Large Language Models (LLMs) for Arabic remain relatively limited. Most existing Arabic benchmarks focus on linguistic, cultural, or religious content, leaving a significant gap in domains like STEM and code which are increasingly relevant for real-world LLM applications. To help bridge this gap, we present 3LM, a suite of three benchmarks designed specifically for Arabic. The first is a set of STEM-related question-answer pairs, naturally sourced from Arabic textbooks and educational worksheets"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.15850","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-07-21T17:58:27Z","cross_cats_sorted":[],"title_canon_sha256":"12778d95beea297d1d9f25e44aa6a295698275a755c9bf8e7ea4c3bcc09d692e","abstract_canon_sha256":"24e8566e160885bddc981aa7eaa844be9dbb3033566ec1e44fb0c1842ea92cad"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:43:14.066321Z","signature_b64":"sR8iWumWVYWSP6/CZduzWz3LJ/w6nSeFZq5dGxZkDShfBebULzfb7MA5wirqmGetpnLWPyZz+BO17BS6fYmQBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d1dc28303ca271a8bc7ee3e943ae8e83c451b846df51ef2e72de545773534da3","last_reissued_at":"2026-07-05T11:43:14.065844Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:43:14.065844Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"3LM: Bridging Arabic, STEM, and Code through Benchmarking","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ahmed Alzubaidi, Basma El Amel Boussaha, Giulia Campesan, Hakim Hacid, Leen AlQadi, Mohammed Alyafeai, Mugariya Farooq, Shaikha Alsuwaidi","submitted_at":"2025-07-21T17:58:27Z","abstract_excerpt":"Arabic is one of the most widely spoken languages in the world, yet efforts to develop and evaluate Large Language Models (LLMs) for Arabic remain relatively limited. Most existing Arabic benchmarks focus on linguistic, cultural, or religious content, leaving a significant gap in domains like STEM and code which are increasingly relevant for real-world LLM applications. To help bridge this gap, we present 3LM, a suite of three benchmarks designed specifically for Arabic. The first is a set of STEM-related question-answer pairs, naturally sourced from Arabic textbooks and educational worksheets"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.15850","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.15850/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.15850","created_at":"2026-07-05T11:43:14.065904+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.15850v3","created_at":"2026-07-05T11:43:14.065904+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.15850","created_at":"2026-07-05T11:43:14.065904+00:00"},{"alias_kind":"pith_short_12","alias_value":"2HOCQMB4UJY2","created_at":"2026-07-05T11:43:14.065904+00:00"},{"alias_kind":"pith_short_16","alias_value":"2HOCQMB4UJY2RPD6","created_at":"2026-07-05T11:43:14.065904+00:00"},{"alias_kind":"pith_short_8","alias_value":"2HOCQMB4","created_at":"2026-07-05T11:43:14.065904+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2HOCQMB4UJY2RPD64PUUHLUOQP","json":"https://pith.science/pith/2HOCQMB4UJY2RPD64PUUHLUOQP.json","graph_json":"https://pith.science/api/pith-number/2HOCQMB4UJY2RPD64PUUHLUOQP/graph.json","events_json":"https://pith.science/api/pith-number/2HOCQMB4UJY2RPD64PUUHLUOQP/events.json","paper":"https://pith.science/paper/2HOCQMB4"},"agent_actions":{"view_html":"https://pith.science/pith/2HOCQMB4UJY2RPD64PUUHLUOQP","download_json":"https://pith.science/pith/2HOCQMB4UJY2RPD64PUUHLUOQP.json","view_paper":"https://pith.science/paper/2HOCQMB4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.15850&json=true","fetch_graph":"https://pith.science/api/pith-number/2HOCQMB4UJY2RPD64PUUHLUOQP/graph.json","fetch_events":"https://pith.science/api/pith-number/2HOCQMB4UJY2RPD64PUUHLUOQP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2HOCQMB4UJY2RPD64PUUHLUOQP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2HOCQMB4UJY2RPD64PUUHLUOQP/action/storage_attestation","attest_author":"https://pith.science/pith/2HOCQMB4UJY2RPD64PUUHLUOQP/action/author_attestation","sign_citation":"https://pith.science/pith/2HOCQMB4UJY2RPD64PUUHLUOQP/action/citation_signature","submit_replication":"https://pith.science/pith/2HOCQMB4UJY2RPD64PUUHLUOQP/action/replication_record"}},"created_at":"2026-07-05T11:43:14.065904+00:00","updated_at":"2026-07-05T11:43:14.065904+00:00"}