{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:CLRQDII65HV6S2W267ZN5T5CUH","short_pith_number":"pith:CLRQDII6","schema_version":"1.0","canonical_sha256":"12e301a11ee9ebe96adaf7f2decfa2a1e27fd17637e8b12c86666bdaf233c363","source":{"kind":"arxiv","id":"2502.06559","version":2},"attestation_state":"computed","paper":{"title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Arman Noroozian, David Fernandez-Llorca, Emilia Gomez, Erasmo Purificato, Guillaume Chaslot, Joao Vinagre, Maria Eriksson","submitted_at":"2025-02-10T15:25:06Z","abstract_excerpt":"Quantitative Artificial Intelligence (AI) Benchmarks have emerged as fundamental tools for evaluating the performance, capability, and safety of AI models and systems. Currently, they shape the direction of AI development and are playing an increasingly prominent role in regulatory frameworks. As their influence grows, however, so too does concerns about how and with what effects they evaluate highly sensitive topics such as capabilities, including high-impact capabilities, safety and systemic risks. This paper presents an interdisciplinary meta-review of about 100 studies that discuss shortco"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.06559","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2025-02-10T15:25:06Z","cross_cats_sorted":[],"title_canon_sha256":"b4c7dc9b507879e966efcdbb6428e87c20f37beea026645f1649a1155e9447ab","abstract_canon_sha256":"6359c7f784125214322cbe1968992d9f40a975daac0aba5aabb0af684a570cae"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:09:07.929024Z","signature_b64":"cJNs5F5fW5hobBrPI7YP7FJuNlF7N7v2T81Xp7Wvc/TgYED2/3NBI3Y0PLJD+kwt+4H6Wvkr3raWGO0YSnNjCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"12e301a11ee9ebe96adaf7f2decfa2a1e27fd17637e8b12c86666bdaf233c363","last_reissued_at":"2026-07-05T11:09:07.928541Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:09:07.928541Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Arman Noroozian, David Fernandez-Llorca, Emilia Gomez, Erasmo Purificato, Guillaume Chaslot, Joao Vinagre, Maria Eriksson","submitted_at":"2025-02-10T15:25:06Z","abstract_excerpt":"Quantitative Artificial Intelligence (AI) Benchmarks have emerged as fundamental tools for evaluating the performance, capability, and safety of AI models and systems. Currently, they shape the direction of AI development and are playing an increasingly prominent role in regulatory frameworks. As their influence grows, however, so too does concerns about how and with what effects they evaluate highly sensitive topics such as capabilities, including high-impact capabilities, safety and systemic risks. This paper presents an interdisciplinary meta-review of about 100 studies that discuss shortco"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.06559","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.06559/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.06559","created_at":"2026-07-05T11:09:07.928599+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.06559v2","created_at":"2026-07-05T11:09:07.928599+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.06559","created_at":"2026-07-05T11:09:07.928599+00:00"},{"alias_kind":"pith_short_12","alias_value":"CLRQDII65HV6","created_at":"2026-07-05T11:09:07.928599+00:00"},{"alias_kind":"pith_short_16","alias_value":"CLRQDII65HV6S2W2","created_at":"2026-07-05T11:09:07.928599+00:00"},{"alias_kind":"pith_short_8","alias_value":"CLRQDII6","created_at":"2026-07-05T11:09:07.928599+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":19,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18158","citing_title":"The Measurement Gap in the Automation of EU Law: Benchmarking Doctrinal Legal Reasoning under the EU AI Act","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09118","citing_title":"ComplexConstraints and Beyond: Expert Rubrics for RLVR","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07996","citing_title":"MC-PDD: Masked Corpus-Level Pretraining Data Detection for Black-Box Large Language Models","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31755","citing_title":"A Technical Typology of AI Systems in Public Administration","ref_index":282,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30916","citing_title":"Welfare, Improvability, and Variance: A Principal-Agent Approach to Optimal Benchmark Item Aggregation","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2602.13372","citing_title":"MoralityGym: A Benchmark for Evaluating Hierarchical Moral Alignment in Sequential Decision-Making Agents","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2603.14987","citing_title":"Beyond Benchmark Islands: Toward Representative Trustworthiness Evaluation for Agentic AI","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2509.19590","citing_title":"Position: AI Evaluations Should be Grounded on a Theory of Capability","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2511.19115","citing_title":"AI Consciousness and Existential Risk","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02406","citing_title":"Evaluating AI-Generated Images of Cultural Artifacts with Community-Informed Rubrics","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2510.15297","citing_title":"VERA-MH Concept Paper","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2602.18911","citing_title":"From Human-Level AI Tales to AI Leveling Human Scales","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14164","citing_title":"Unsteady Metrics and Benchmarking Cultures of AI Model Builders","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16403","citing_title":"Computational Hermeneutics: Evaluating generative AI as a cultural technology","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2604.01375","citing_title":"RIFT: A RubrIc Failure Mode Taxonomy and Automated Diagnostics","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02406","citing_title":"Evaluating AI-Generated Images of Cultural Artifacts with Community-Informed Rubrics","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06865","citing_title":"Dataset Watermarking for Closed LLMs with Provable Detection","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05274","citing_title":"Simulating the Evolution of Alignment and Values in Machine Intelligence","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.28053","citing_title":"To Build or Not to Build? Factors that Lead to Non-Development or Abandonment of AI Systems","ref_index":44,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CLRQDII65HV6S2W267ZN5T5CUH","json":"https://pith.science/pith/CLRQDII65HV6S2W267ZN5T5CUH.json","graph_json":"https://pith.science/api/pith-number/CLRQDII65HV6S2W267ZN5T5CUH/graph.json","events_json":"https://pith.science/api/pith-number/CLRQDII65HV6S2W267ZN5T5CUH/events.json","paper":"https://pith.science/paper/CLRQDII6"},"agent_actions":{"view_html":"https://pith.science/pith/CLRQDII65HV6S2W267ZN5T5CUH","download_json":"https://pith.science/pith/CLRQDII65HV6S2W267ZN5T5CUH.json","view_paper":"https://pith.science/paper/CLRQDII6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.06559&json=true","fetch_graph":"https://pith.science/api/pith-number/CLRQDII65HV6S2W267ZN5T5CUH/graph.json","fetch_events":"https://pith.science/api/pith-number/CLRQDII65HV6S2W267ZN5T5CUH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CLRQDII65HV6S2W267ZN5T5CUH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CLRQDII65HV6S2W267ZN5T5CUH/action/storage_attestation","attest_author":"https://pith.science/pith/CLRQDII65HV6S2W267ZN5T5CUH/action/author_attestation","sign_citation":"https://pith.science/pith/CLRQDII65HV6S2W267ZN5T5CUH/action/citation_signature","submit_replication":"https://pith.science/pith/CLRQDII65HV6S2W267ZN5T5CUH/action/replication_record"}},"created_at":"2026-07-05T11:09:07.928599+00:00","updated_at":"2026-07-05T11:09:07.928599+00:00"}