{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:3R6CHEEBKP6ICBFN44UYAL2T5Y","short_pith_number":"pith:3R6CHEEB","schema_version":"1.0","canonical_sha256":"dc7c23908153fc8104ade729802f53ee3e4193731bad2a7668d357395900380a","source":{"kind":"arxiv","id":"2507.20419","version":1},"attestation_state":"computed","paper":{"title":"Survey of NLU Benchmarks Diagnosing Linguistic Phenomena: Why not Standardize Diagnostics Benchmarks?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.HC","cs.LG"],"primary_cat":"cs.CL","authors_text":"Ghaida Rebdawi, Khloud Al Jallad, Nada Ghneim","submitted_at":"2025-07-27T21:30:50Z","abstract_excerpt":"Natural Language Understanding (NLU) is a basic task in Natural Language Processing (NLP). The evaluation of NLU capabilities has become a trending research topic that attracts researchers in the last few years, resulting in the development of numerous benchmarks. These benchmarks include various tasks and datasets in order to evaluate the results of pretrained models via public leaderboards. Notably, several benchmarks contain diagnostics datasets designed for investigation and fine-grained error analysis across a wide range of linguistic phenomena. This survey provides a comprehensive review"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.20419","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-07-27T21:30:50Z","cross_cats_sorted":["cs.AI","cs.HC","cs.LG"],"title_canon_sha256":"d4d244454f8c7cdfeb7e59c2b95042397bebf55a6cdc89ea5c62a40ec1ee9608","abstract_canon_sha256":"86563939bf6345df27d2a3683443085dda17f66a6799f256c142f9acf038e022"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:44:19.115770Z","signature_b64":"PaLVDxzqbKcV4288N3wGkbVEWNCq6neCBvvnCwwveWZ0z9qhBDIfoPpcq7rlRfSnMVeg7MlLd4w41eR9J+MjDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dc7c23908153fc8104ade729802f53ee3e4193731bad2a7668d357395900380a","last_reissued_at":"2026-07-05T11:44:19.115304Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:44:19.115304Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Survey of NLU Benchmarks Diagnosing Linguistic Phenomena: Why not Standardize Diagnostics Benchmarks?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.HC","cs.LG"],"primary_cat":"cs.CL","authors_text":"Ghaida Rebdawi, Khloud Al Jallad, Nada Ghneim","submitted_at":"2025-07-27T21:30:50Z","abstract_excerpt":"Natural Language Understanding (NLU) is a basic task in Natural Language Processing (NLP). The evaluation of NLU capabilities has become a trending research topic that attracts researchers in the last few years, resulting in the development of numerous benchmarks. These benchmarks include various tasks and datasets in order to evaluate the results of pretrained models via public leaderboards. Notably, several benchmarks contain diagnostics datasets designed for investigation and fine-grained error analysis across a wide range of linguistic phenomena. This survey provides a comprehensive review"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.20419","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.20419/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.20419","created_at":"2026-07-05T11:44:19.115368+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.20419v1","created_at":"2026-07-05T11:44:19.115368+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.20419","created_at":"2026-07-05T11:44:19.115368+00:00"},{"alias_kind":"pith_short_12","alias_value":"3R6CHEEBKP6I","created_at":"2026-07-05T11:44:19.115368+00:00"},{"alias_kind":"pith_short_16","alias_value":"3R6CHEEBKP6ICBFN","created_at":"2026-07-05T11:44:19.115368+00:00"},{"alias_kind":"pith_short_8","alias_value":"3R6CHEEB","created_at":"2026-07-05T11:44:19.115368+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3R6CHEEBKP6ICBFN44UYAL2T5Y","json":"https://pith.science/pith/3R6CHEEBKP6ICBFN44UYAL2T5Y.json","graph_json":"https://pith.science/api/pith-number/3R6CHEEBKP6ICBFN44UYAL2T5Y/graph.json","events_json":"https://pith.science/api/pith-number/3R6CHEEBKP6ICBFN44UYAL2T5Y/events.json","paper":"https://pith.science/paper/3R6CHEEB"},"agent_actions":{"view_html":"https://pith.science/pith/3R6CHEEBKP6ICBFN44UYAL2T5Y","download_json":"https://pith.science/pith/3R6CHEEBKP6ICBFN44UYAL2T5Y.json","view_paper":"https://pith.science/paper/3R6CHEEB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.20419&json=true","fetch_graph":"https://pith.science/api/pith-number/3R6CHEEBKP6ICBFN44UYAL2T5Y/graph.json","fetch_events":"https://pith.science/api/pith-number/3R6CHEEBKP6ICBFN44UYAL2T5Y/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3R6CHEEBKP6ICBFN44UYAL2T5Y/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3R6CHEEBKP6ICBFN44UYAL2T5Y/action/storage_attestation","attest_author":"https://pith.science/pith/3R6CHEEBKP6ICBFN44UYAL2T5Y/action/author_attestation","sign_citation":"https://pith.science/pith/3R6CHEEBKP6ICBFN44UYAL2T5Y/action/citation_signature","submit_replication":"https://pith.science/pith/3R6CHEEBKP6ICBFN44UYAL2T5Y/action/replication_record"}},"created_at":"2026-07-05T11:44:19.115368+00:00","updated_at":"2026-07-05T11:44:19.115368+00:00"}