{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:HW2WPJIF4SNDXBWZ6KZZO7WQAL","short_pith_number":"pith:HW2WPJIF","schema_version":"1.0","canonical_sha256":"3db567a505e49a3b86d9f2b3977ed002cc86a74ca6568503a84056910a7422e4","source":{"kind":"arxiv","id":"2505.10573","version":4},"attestation_state":"computed","paper":{"title":"Measurement to Meaning: A Validity-Centered Framework for AI Evaluation","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CY","authors_text":"Ahmed Ahmed, Angelina Wang, Anka Reuel, Ben Domingue, Olawale Salaudeen, Sanmi Koyejo, Sudharsan Sundar, Suhana Bedi, Zachary Robertson","submitted_at":"2025-05-13T20:36:22Z","abstract_excerpt":"While the capabilities and utility of AI systems have advanced, rigorous norms for evaluating these systems have lagged. Grand claims, such as models achieving general reasoning capabilities, are supported with model performance on narrow benchmarks, like performance on graduate-level exam questions, which provide a limited and potentially misleading assessment. We provide a structured approach for reasoning about the types of evaluative claims that can be made given the available evidence. For instance, our framework helps determine whether performance on a mathematical benchmark is an indica"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.10573","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CY","submitted_at":"2025-05-13T20:36:22Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"a1d203a0cb2e670bd9b64e621351ff83824777b854eddc19a6481da240a94dac","abstract_canon_sha256":"fa785fcb6a43362a953da4a55b543f514da11fa631bd2319975a37fd9a7e2968"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:27:30.083031Z","signature_b64":"RcRXbk4BFG4KKw9jbQjlomO5s8tcTISAdfcBoNk3mTj/DvxqLr9FpMp9wxyU9Fa2tqusjwEbrg5sYOGJZaIYDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3db567a505e49a3b86d9f2b3977ed002cc86a74ca6568503a84056910a7422e4","last_reissued_at":"2026-07-05T11:27:30.082541Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:27:30.082541Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Measurement to Meaning: A Validity-Centered Framework for AI Evaluation","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CY","authors_text":"Ahmed Ahmed, Angelina Wang, Anka Reuel, Ben Domingue, Olawale Salaudeen, Sanmi Koyejo, Sudharsan Sundar, Suhana Bedi, Zachary Robertson","submitted_at":"2025-05-13T20:36:22Z","abstract_excerpt":"While the capabilities and utility of AI systems have advanced, rigorous norms for evaluating these systems have lagged. Grand claims, such as models achieving general reasoning capabilities, are supported with model performance on narrow benchmarks, like performance on graduate-level exam questions, which provide a limited and potentially misleading assessment. We provide a structured approach for reasoning about the types of evaluative claims that can be made given the available evidence. For instance, our framework helps determine whether performance on a mathematical benchmark is an indica"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.10573","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.10573/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.10573","created_at":"2026-07-05T11:27:30.082598+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.10573v4","created_at":"2026-07-05T11:27:30.082598+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.10573","created_at":"2026-07-05T11:27:30.082598+00:00"},{"alias_kind":"pith_short_12","alias_value":"HW2WPJIF4SND","created_at":"2026-07-05T11:27:30.082598+00:00"},{"alias_kind":"pith_short_16","alias_value":"HW2WPJIF4SNDXBWZ","created_at":"2026-07-05T11:27:30.082598+00:00"},{"alias_kind":"pith_short_8","alias_value":"HW2WPJIF","created_at":"2026-07-05T11:27:30.082598+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26079","citing_title":"Same Evidence, Different Answer: Auditing Order Sensitivity in Multimodal Large Language Models","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10154","citing_title":"Quality Is Not a Safety Proxy Under Quantization","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30916","citing_title":"Welfare, Improvability, and Variance: A Principal-Agent Approach to Optimal Benchmark Item Aggregation","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02406","citing_title":"Evaluating AI-Generated Images of Cultural Artifacts with Community-Informed Rubrics","ref_index":99,"is_internal_anchor":false},{"citing_arxiv_id":"2509.14528","citing_title":"Why Johnny Can't Use Agents: Industry Aspirations vs. User Realities with AI Agents","ref_index":67,"is_internal_anchor":false},{"citing_arxiv_id":"2602.17753","citing_title":"The 2025 AI Agent Index: Documenting Technical and Safety Features of Deployed Agentic AI Systems","ref_index":113,"is_internal_anchor":false},{"citing_arxiv_id":"2603.06811","citing_title":"Making AI Evaluation Deployment Relevant Through Context Specification","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02406","citing_title":"Evaluating AI-Generated Images of Cultural Artifacts with Community-Informed Rubrics","ref_index":99,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19357","citing_title":"FairTree: Subgroup Fairness Auditing of Machine Learning Models with Bias-Variance Decomposition","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06652","citing_title":"When No Benchmark Exists: Validating Comparative LLM Safety Scoring Without Ground-Truth Labels","ref_index":27,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HW2WPJIF4SNDXBWZ6KZZO7WQAL","json":"https://pith.science/pith/HW2WPJIF4SNDXBWZ6KZZO7WQAL.json","graph_json":"https://pith.science/api/pith-number/HW2WPJIF4SNDXBWZ6KZZO7WQAL/graph.json","events_json":"https://pith.science/api/pith-number/HW2WPJIF4SNDXBWZ6KZZO7WQAL/events.json","paper":"https://pith.science/paper/HW2WPJIF"},"agent_actions":{"view_html":"https://pith.science/pith/HW2WPJIF4SNDXBWZ6KZZO7WQAL","download_json":"https://pith.science/pith/HW2WPJIF4SNDXBWZ6KZZO7WQAL.json","view_paper":"https://pith.science/paper/HW2WPJIF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.10573&json=true","fetch_graph":"https://pith.science/api/pith-number/HW2WPJIF4SNDXBWZ6KZZO7WQAL/graph.json","fetch_events":"https://pith.science/api/pith-number/HW2WPJIF4SNDXBWZ6KZZO7WQAL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HW2WPJIF4SNDXBWZ6KZZO7WQAL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HW2WPJIF4SNDXBWZ6KZZO7WQAL/action/storage_attestation","attest_author":"https://pith.science/pith/HW2WPJIF4SNDXBWZ6KZZO7WQAL/action/author_attestation","sign_citation":"https://pith.science/pith/HW2WPJIF4SNDXBWZ6KZZO7WQAL/action/citation_signature","submit_replication":"https://pith.science/pith/HW2WPJIF4SNDXBWZ6KZZO7WQAL/action/replication_record"}},"created_at":"2026-07-05T11:27:30.082598+00:00","updated_at":"2026-07-05T11:27:30.082598+00:00"}