{"as_of":"2026-08-23T17:11:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:57b6c712866008989b7f0455d3b7a48e9c327a1536be1929bbdc8e9a7b8766d1","coverage":[{"denominator":135,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T15:06:55.228037Z","state":"measured"},{"denominator":133,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":133,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-23T06:30:58.430688+00:00","state":"measured"},{"denominator":33,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":33,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T11:34:50.154288Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":9,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.06559","snapshot_observed_at":"2026-08-07T11:34:50.154288Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.01920","last_updated":"2025-06-02T17:39:50Z","snapshot_observed_at":"2026-08-13T14:50:56.110277Z","submitted_at":"2025-06-02T17:39:50Z","title":"From Guidelines to Practice: A New Paradigm for Arabic Language Model Evaluation","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-07T11:34:50.154288Z"},"links":{"cited_paper":"/paper/2502.06559","citing_paper":"/paper/2506.01920"},"observation_digest":"sha256:7e63c89b22dd9b56bc20108d43e096b29fa6a1e59a96c0ac97b9842db1c5aca8","observation_id":"0820906a-b442-4c1b-96fa-127cc3db3aa2","resolution":{"observed_at":"2026-08-07T11:34:50.154288Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.06559","snapshot_observed_at":"2026-08-07T04:06:28.735359Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.11604","last_updated":"2025-06-27T10:12:42Z","snapshot_observed_at":"2026-08-14T20:56:09.706610Z","submitted_at":"2025-06-13T09:20:41Z","title":"VLM@school -- Evaluation of AI image understanding on German middle school knowledge","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T04:06:28.735359Z"},"links":{"cited_paper":"/paper/2502.06559","citing_paper":"/paper/2506.11604"},"observation_digest":"sha256:681d11d151e0af393aa712145cb492ae18707e501c277f9a3e9a82a75e6bbfd3","observation_id":"06e3eb89-bf89-47cc-b678-bfc63a1c8353","resolution":{"observed_at":"2026-08-07T04:06:28.735359Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.06559","snapshot_observed_at":"2026-08-06T23:25:46.923418Z","title":"Can we trust ai benchmarks? an interdisciplinary review of current issues in ai evaluation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.18213","last_updated":"2025-06-23T00:19:27Z","snapshot_observed_at":"2026-08-16T11:32:18.447851Z","submitted_at":"2025-06-23T00:19:27Z","title":"A Conceptual Framework for AI Capability Evaluations","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-06T23:25:46.923418Z"},"links":{"cited_paper":"/paper/2502.06559","citing_paper":"/paper/2506.18213"},"observation_digest":"sha256:f0304e7fb0a8dc32417f9f7673569900978a0e122411ffee44622fd3119a5f5c","observation_id":"bd68057e-87f1-4259-aafc-0a5e2ca81e00","resolution":{"observed_at":"2026-08-06T23:25:46.923418Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.06559","snapshot_observed_at":"2026-08-06T21:45:01.187973Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation , 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.23706","last_updated":"2025-06-30T10:29:42Z","snapshot_observed_at":"2026-08-22T15:19:08.990607Z","submitted_at":"2025-06-30T10:29:42Z","title":"Attestable Audits: Verifiable AI Safety Benchmarks Using Trusted Execution Environments","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-06T21:45:01.187973Z"},"links":{"cited_paper":"/paper/2502.06559","citing_paper":"/paper/2506.23706"},"observation_digest":"sha256:e343f9b39147b102fa89f11a91034f7d3ca8bc2cf2c844c29b4d44b8b1e2d4bd","observation_id":"bfb8a087-d194-4022-99b5-ae7bfe110e16","resolution":{"observed_at":"2026-08-06T21:45:01.187973Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.06559","snapshot_observed_at":"2026-08-06T20:24:22.007624Z","title":"Can we trust ai benchmarks? an interdisciplinary review of current issues in ai evaluation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-22T09:27:33.969418Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:22.007624Z"},"links":{"cited_paper":"/paper/2502.06559","citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:708557a8f7a267f378518f564c88b2729cb4d9aea8afc35595b5cef34bf79233","observation_id":"06be70c0-2a6c-4b6d-9947-589ce3f0d3ab","resolution":{"observed_at":"2026-08-06T20:24:22.007624Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.06559","snapshot_observed_at":"2026-08-06T19:48:44.495995Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation, May 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.04575","last_updated":"2025-07-06T23:18:51Z","snapshot_observed_at":"2026-08-19T00:35:57.787865Z","submitted_at":"2025-07-06T23:18:51Z","title":"Lilith: Developmental Modular LLMs with Chemical Signaling","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T19:48:44.495995Z"},"links":{"cited_paper":"/paper/2502.06559","citing_paper":"/paper/2507.04575"},"observation_digest":"sha256:ce7dd5be467d89acebe34c5d1836d889c19d5708d4947d43bd99fcd7099945aa","observation_id":"b4a3e686-dfa0-4a06-b22f-22a90cce61ab","resolution":{"observed_at":"2026-08-06T19:48:44.495995Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.06559","snapshot_observed_at":"2026-08-06T19:07:40.423706Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.06434","last_updated":"2025-07-08T22:29:06Z","snapshot_observed_at":"2026-08-19T16:15:23.665750Z","submitted_at":"2025-07-08T22:29:06Z","title":"Deprecating Benchmarks: Criteria and Framework","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-06T19:07:40.423706Z"},"links":{"cited_paper":"/paper/2502.06559","citing_paper":"/paper/2507.06434"},"observation_digest":"sha256:472287a18fdc7617c6bb1d4799a5dac785ff1a26af0574701860beceba419565","observation_id":"107adcf9-e699-4709-903b-29ddec1b7d1d","resolution":{"observed_at":"2026-08-06T19:07:40.423706Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.06559","snapshot_observed_at":"2026-08-06T13:54:38.308630Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.20014","last_updated":"2025-07-30T08:46:55Z","snapshot_observed_at":"2026-08-15T17:09:56.289108Z","submitted_at":"2025-07-26T17:07:01Z","title":"Policy-Driven AI in Dataspaces: Taxonomy, Explainability, and Pathways for Compliant Innovation","version":2},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-06T13:54:38.308630Z"},"links":{"cited_paper":"/paper/2502.06559","citing_paper":"/paper/2507.20014"},"observation_digest":"sha256:1c63dfc8d79feba852d7f7a6a00d6117f8a09e5d201c097b074c00cf1c4ceedd","observation_id":"09d9021e-42b6-4575-a7eb-8b70337a44f9","resolution":{"observed_at":"2026-08-06T13:54:38.308630Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.06559","snapshot_observed_at":"2026-08-05T20:29:42.978782Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.10358","last_updated":"2025-08-14T05:55:42Z","snapshot_observed_at":"2026-08-14T21:19:15.435784Z","submitted_at":"2025-08-14T05:55:42Z","title":"What to Ask Next? Probing the Imaginative Reasoning of LLMs with TurtleSoup Puzzles","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-05T20:29:42.978782Z"},"links":{"cited_paper":"/paper/2502.06559","citing_paper":"/paper/2508.10358"},"observation_digest":"sha256:4526faee7d95b83b4a920cc1858c1b4a6cc6d5e87ba1e1e45d6b4bf4e3b61dd1","observation_id":"90985508-fb8d-4c1d-8918-c2a5a016e410","resolution":{"observed_at":"2026-08-05T20:29:42.978782Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"cited_work":{"arxiv_id":"2502.06559","doi":"10.48550/arxiv.2502.06559","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.06559","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Can we trust ai benchmarks? an interdisciplinary review of current issues in ai evaluation","venue":"ArXiv.org","work_id":"cdcf1d27-8950-4301-b2c5-4fa9aad92cab","year":2025},"citing_paper":{"arxiv_id":"2509.19590","last_updated":"2026-05-16T01:07:51Z","snapshot_observed_at":"2026-08-16T22:43:15.745999Z","submitted_at":"2025-09-23T21:29:04Z","title":"Position: AI Evaluations Should be Grounded on a Theory of Capability","version":2},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-05-21T21:57:55.834632Z"},"links":{"cited_paper":"/paper/2502.06559","citing_paper":"/paper/2509.19590"},"observation_digest":"sha256:dac1572dacd02d6e2c2886cf2712bb2f99022c2d0e8a418c63380defc42051b2","observation_id":"1f9b9a06-7551-48c2-b207-11eb412f0312","resolution":{"observed_at":"2026-05-21T22:00:41.457894Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-04T01:08:05.510513+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-04T01:08:05.510513+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"cited_work":{"arxiv_id":"2502.06559","doi":"10.48550/arxiv.2502.06559","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.06559","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Can we trust ai benchmarks? an interdisciplinary review of current issues in ai evaluation","venue":"ArXiv.org","work_id":"cdcf1d27-8950-4301-b2c5-4fa9aad92cab","year":2025},"citing_paper":{"arxiv_id":"2510.15297","last_updated":"2026-05-13T10:08:20Z","snapshot_observed_at":"2026-08-15T15:34:32.435246Z","submitted_at":"2025-10-17T04:07:29Z","title":"VERA-MH Concept Paper","version":4},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-18T07:00:15.411310Z"},"links":{"cited_paper":"/paper/2502.06559","citing_paper":"/paper/2510.15297"},"observation_digest":"sha256:5a96329e9a2b227c3626c3fac347f2183a3db883b3af499ad6b8a12d9efea337","observation_id":"0bb11615-6862-4f1c-8fdd-7e5f22ddc3e1","resolution":{"observed_at":"2026-05-18T07:01:01.475860Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-04T01:08:05.510513+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-04T01:08:05.510513+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"cited_work":{"arxiv_id":"2502.06559","doi":"10.48550/arxiv.2502.06559","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.06559","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Can we trust ai benchmarks? an interdisciplinary review of current issues in ai evaluation","venue":"ArXiv.org","work_id":"cdcf1d27-8950-4301-b2c5-4fa9aad92cab","year":2025},"citing_paper":{"arxiv_id":"2511.19115","last_updated":"2026-05-14T19:57:37Z","snapshot_observed_at":"2026-08-14T12:59:48.878739Z","submitted_at":"2025-11-24T13:48:02Z","title":"AI Consciousness and Existential Risk","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-21T18:26:03.464658Z"},"links":{"cited_paper":"/paper/2502.06559","citing_paper":"/paper/2511.19115"},"observation_digest":"sha256:3d3100a2ea3b346f3dcd18f07f6d36a94f532b6fe5c3745fd30b24c5bfa5128f","observation_id":"403e1f22-2bd4-4867-9dbb-ea5ae6e3947c","resolution":{"observed_at":"2026-05-21T18:30:29.085892Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-04T01:08:05.510513+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-04T01:08:05.510513+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.06559","snapshot_observed_at":"2026-08-03T04:24:17.550034Z","title":"Can we trust AI benchmarks? An interdisciplinary review of current issues in AI evaluation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.05088","last_updated":"2026-07-07T19:40:51Z","snapshot_observed_at":"2026-08-19T17:36:06.822243Z","submitted_at":"2026-02-04T22:17:04Z","title":"AI Chatbot Suicide Risk Detection and Response: Human Validation Study of the Open-Source VERA-MH Safety Evaluation","version":4},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-03T04:24:17.550034Z"},"links":{"cited_paper":"/paper/2502.06559","citing_paper":"/paper/2602.05088"},"observation_digest":"sha256:08390602256d646b4f3f3ea1d369867bf63cd5bb1fac2679c5b53c99dadcb44e","observation_id":"85ada9e8-d58f-4c86-9f18-21b1d4774eda","resolution":{"observed_at":"2026-08-03T04:24:17.550034Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"cited_work":{"arxiv_id":"2502.06559","doi":"10.48550/arxiv.2502.06559","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.06559","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Can we trust ai benchmarks? an interdisciplinary review of current issues in ai evaluation","venue":"ArXiv.org","work_id":"cdcf1d27-8950-4301-b2c5-4fa9aad92cab","year":2025},"citing_paper":{"arxiv_id":"2602.13372","last_updated":"2026-05-21T13:32:14Z","snapshot_observed_at":"2026-08-18T17:11:29.379708Z","submitted_at":"2026-02-13T15:40:32Z","title":"MoralityGym: A Benchmark for Evaluating Hierarchical Moral Alignment in Sequential Decision-Making Agents","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-22T10:49:35.846590Z"},"links":{"cited_paper":"/paper/2502.06559","citing_paper":"/paper/2602.13372"},"observation_digest":"sha256:be8fa6ba9d467949a2d31d8292a303859a53e9b54ffbce8fe9e3d41bef58feba","observation_id":"07cf78a3-31ca-40b9-9cc0-961e5a657f95","resolution":{"observed_at":"2026-05-22T10:51:25.927108Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-04T01:08:05.510513+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-04T01:08:05.510513+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"cited_work":{"arxiv_id":"2502.06559","doi":"10.48550/arxiv.2502.06559","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.06559","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Can we trust ai benchmarks? an interdisciplinary review of current issues in ai evaluation","venue":"ArXiv.org","work_id":"cdcf1d27-8950-4301-b2c5-4fa9aad92cab","year":2025},"citing_paper":{"arxiv_id":"2602.18911","last_updated":"2026-04-06T20:01:09Z","snapshot_observed_at":"2026-08-13T08:27:14.711191Z","submitted_at":"2026-02-21T17:27:20Z","title":"From Human-Level AI Tales to AI Leveling Human Scales","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-15T20:08:29.604466Z"},"links":{"cited_paper":"/paper/2502.06559","citing_paper":"/paper/2602.18911"},"observation_digest":"sha256:5c7b2c053fd48e3abca18b8d1fe1fa2aae7b7843697d71009fe58c43eb7ca8b1","observation_id":"234dd87e-67d2-4a20-8fa5-f6a6429e8ff6","resolution":{"observed_at":"2026-05-15T20:10:18.105298Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-04T01:08:05.510513+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-04T01:08:05.510513+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"cited_work":{"arxiv_id":"2502.06559","doi":"10.48550/arxiv.2502.06559","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.06559","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Can we trust ai benchmarks? an interdisciplinary review of current issues in ai evaluation","venue":"ArXiv.org","work_id":"cdcf1d27-8950-4301-b2c5-4fa9aad92cab","year":2025},"citing_paper":{"arxiv_id":"2603.14987","last_updated":"2026-05-21T06:24:06Z","snapshot_observed_at":"2026-08-03T21:30:30.993382Z","submitted_at":"2026-03-16T08:51:33Z","title":"Beyond Benchmark Islands: Toward Representative Trustworthiness Evaluation for Agentic AI","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-22T10:19:56.003219Z"},"links":{"cited_paper":"/paper/2502.06559","citing_paper":"/paper/2603.14987"},"observation_digest":"sha256:c33b92d5cf15970bf18f3cf16188381b1911620701fcc4774ccd64f905c4f3dd","observation_id":"f67720ca-b7d7-4365-871d-f636c17c07a5","resolution":{"observed_at":"2026-05-22T10:21:23.982835Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-04T01:08:05.510513+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-04T01:08:05.510513+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"cited_work":{"arxiv_id":"2502.06559","doi":"10.48550/arxiv.2502.06559","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.06559","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Can we trust ai benchmarks? an interdisciplinary review of current issues in ai evaluation","venue":"ArXiv.org","work_id":"cdcf1d27-8950-4301-b2c5-4fa9aad92cab","year":2025},"citing_paper":{"arxiv_id":"2604.01375","last_updated":"2026-04-20T23:25:41Z","snapshot_observed_at":"2026-08-14T14:29:03.603297Z","submitted_at":"2026-04-01T20:34:43Z","title":"RIFT: A RubrIc Failure Mode Taxonomy and Automated Diagnostics","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-13T22:25:35.345087Z"},"links":{"cited_paper":"/paper/2502.06559","citing_paper":"/paper/2604.01375"},"observation_digest":"sha256:34fb0a033167aa2012b3499af5c98355617137a062c55d72f613bd062806f7e8","observation_id":"eb409c71-7bca-4f56-958b-84895575230e","resolution":{"observed_at":"2026-05-13T22:28:21.220651Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-04T01:08:05.510513+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-04T01:08:05.510513+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"cited_work":{"arxiv_id":"2502.06559","doi":"10.48550/arxiv.2502.06559","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.06559","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Can we trust ai benchmarks? an interdisciplinary review of current issues in ai evaluation","venue":"ArXiv.org","work_id":"cdcf1d27-8950-4301-b2c5-4fa9aad92cab","year":2025},"citing_paper":{"arxiv_id":"2604.02406","last_updated":"2026-05-16T00:42:12Z","snapshot_observed_at":"2026-08-21T09:02:10.537989Z","submitted_at":"2026-04-02T17:17:12Z","title":"Evaluating AI-Generated Images of Cultural Artifacts with Community-Informed Rubrics","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-13T20:59:52.448832Z"},"links":{"cited_paper":"/paper/2502.06559","citing_paper":"/paper/2604.02406"},"observation_digest":"sha256:c7865bc17771fec33fd08f9826ef6b49ef383a0688f224ac8cdcfb0eaa4953f8","observation_id":"15b4690f-8238-4d2f-a9e6-4ef43fdc3c52","resolution":{"observed_at":"2026-05-13T21:03:20.101226Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-04T01:08:05.510513+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-04T01:08:05.510513+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"cited_work":{"arxiv_id":"2502.06559","doi":"10.48550/arxiv.2502.06559","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.06559","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Can we trust ai benchmarks? an interdisciplinary review of current issues in ai evaluation","venue":"ArXiv.org","work_id":"cdcf1d27-8950-4301-b2c5-4fa9aad92cab","year":2025},"citing_paper":{"arxiv_id":"2604.02406","last_updated":"2026-05-16T00:42:12Z","snapshot_observed_at":"2026-08-21T09:02:10.537989Z","submitted_at":"2026-04-02T17:17:12Z","title":"Evaluating AI-Generated Images of Cultural Artifacts with Community-Informed Rubrics","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-21T10:35:39.269869Z"},"links":{"cited_paper":"/paper/2502.06559","citing_paper":"/paper/2604.02406"},"observation_digest":"sha256:1402d08231e3db1bbb8b1b4779ad6c307f4b632c83e95b597c1e3335a994948c","observation_id":"5ef79f9c-5b4b-4440-a325-701f8b083403","resolution":{"observed_at":"2026-05-21T10:40:00.798502Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-04T01:08:05.510513+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-04T01:08:05.510513+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"cited_work":{"arxiv_id":"2502.06559","doi":"10.48550/arxiv.2502.06559","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.06559","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Can we trust ai benchmarks? an interdisciplinary review of current issues in ai evaluation","venue":"ArXiv.org","work_id":"cdcf1d27-8950-4301-b2c5-4fa9aad92cab","year":2025},"citing_paper":{"arxiv_id":"2604.05274","last_updated":"2026-04-07T00:18:28Z","snapshot_observed_at":"2026-07-06T22:54:04.491386Z","submitted_at":"2026-04-07T00:18:28Z","title":"Simulating the Evolution of Alignment and Values in Machine Intelligence","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-05-10T20:15:46.311347Z"},"links":{"cited_paper":"/paper/2502.06559","citing_paper":"/paper/2604.05274"},"observation_digest":"sha256:4e4dfd464b0cc748fa98674b13d8e1e9e17471a34c01cb26a187ca59616cbc0c","observation_id":"e29ccdeb-27b1-4b9b-b1b4-d1073cd12667","resolution":{"observed_at":"2026-05-10T22:05:48.445792Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-04T01:08:05.510513+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-04T01:08:05.510513+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"cited_work":{"arxiv_id":"2502.06559","doi":"10.48550/arxiv.2502.06559","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.06559","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Can we trust ai benchmarks? an interdisciplinary review of current issues in ai evaluation","venue":"ArXiv.org","work_id":"cdcf1d27-8950-4301-b2c5-4fa9aad92cab","year":2025},"citing_paper":{"arxiv_id":"2604.16403","last_updated":"2026-03-31T12:18:56Z","snapshot_observed_at":"2026-08-13T03:56:07.949173Z","submitted_at":"2026-03-31T12:18:56Z","title":"Computational Hermeneutics: Evaluating generative AI as a cultural technology","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-13T23:48:24.354896Z"},"links":{"cited_paper":"/paper/2502.06559","citing_paper":"/paper/2604.16403"},"observation_digest":"sha256:e38596bb8cdad05309099d1a0281592453316096d05f344268dbb050d56f8ca3","observation_id":"952a38f4-bfb9-4293-b86f-f9f5c3d0001a","resolution":{"observed_at":"2026-05-13T23:48:27.153713Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-04T01:08:05.510513+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-04T01:08:05.510513+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"cited_work":{"arxiv_id":"2502.06559","doi":"10.48550/arxiv.2502.06559","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.06559","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Can we trust ai benchmarks? an interdisciplinary review of current issues in ai evaluation","venue":"ArXiv.org","work_id":"cdcf1d27-8950-4301-b2c5-4fa9aad92cab","year":2025},"citing_paper":{"arxiv_id":"2604.28053","last_updated":"2026-04-30T16:00:52Z","snapshot_observed_at":"2026-08-11T15:41:37.646967Z","submitted_at":"2026-04-30T16:00:52Z","title":"To Build or Not to Build? Factors that Lead to Non-Development or Abandonment of AI Systems","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-07T06:44:14.093749Z"},"links":{"cited_paper":"/paper/2502.06559","citing_paper":"/paper/2604.28053"},"observation_digest":"sha256:385fc13b30955075c5ee6b46cfde0b17c24c747cf6a2b93e2a6c03068a966c7a","observation_id":"e6a615d0-b24d-4a72-ba51-82f817ff86fc","resolution":{"observed_at":"2026-05-09T04:55:11.735752Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-04T01:08:05.510513+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-04T01:08:05.510513+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"cited_work":{"arxiv_id":"2502.06559","doi":"10.48550/arxiv.2502.06559","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.06559","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Can we trust ai benchmarks? an interdisciplinary review of current issues in ai evaluation","venue":"ArXiv.org","work_id":"cdcf1d27-8950-4301-b2c5-4fa9aad92cab","year":2025},"citing_paper":{"arxiv_id":"2605.06865","last_updated":"2026-05-07T19:06:35Z","snapshot_observed_at":"2026-07-06T23:19:15.382283Z","submitted_at":"2026-05-07T19:06:35Z","title":"Dataset Watermarking for Closed LLMs with Provable Detection","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-11T00:53:42.185498Z"},"links":{"cited_paper":"/paper/2502.06559","citing_paper":"/paper/2605.06865"},"observation_digest":"sha256:91884a3b95fd8927bd7cb6510e6c014fd1cfd958ef1fab5fdeb520b913d96c66","observation_id":"2d70d903-9445-4603-a96e-4fc68942de23","resolution":{"observed_at":"2026-05-11T05:00:57.214146Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-04T01:08:05.510513+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-04T01:08:05.510513+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"cited_work":{"arxiv_id":"2502.06559","doi":"10.48550/arxiv.2502.06559","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.06559","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Can we trust ai benchmarks? an interdisciplinary review of current issues in ai evaluation","venue":"ArXiv.org","work_id":"cdcf1d27-8950-4301-b2c5-4fa9aad92cab","year":2025},"citing_paper":{"arxiv_id":"2605.14164","last_updated":"2026-05-13T22:39:10Z","snapshot_observed_at":"2026-08-21T13:51:30.855697Z","submitted_at":"2026-05-13T22:39:10Z","title":"Unsteady Metrics and Benchmarking Cultures of AI Model Builders","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-15T04:54:26.888562Z"},"links":{"cited_paper":"/paper/2502.06559","citing_paper":"/paper/2605.14164"},"observation_digest":"sha256:a51c1723335e98e198d4035358c52c07f85cf809812be2e4b071fb19d4f2dc7c","observation_id":"0226a078-f945-4fc6-8b3e-278471d5bfc0","resolution":{"observed_at":"2026-05-15T04:55:03.336380Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-04T01:08:05.510513+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-04T01:08:05.510513+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"cited_work":{"arxiv_id":"2502.06559","doi":"10.48550/arxiv.2502.06559","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.06559","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Can we trust ai benchmarks? an interdisciplinary review of current issues in ai evaluation","venue":"ArXiv.org","work_id":"cdcf1d27-8950-4301-b2c5-4fa9aad92cab","year":2025},"citing_paper":{"arxiv_id":"2605.30916","last_updated":"2026-05-29T07:01:38Z","snapshot_observed_at":"2026-08-06T11:05:12.049404Z","submitted_at":"2026-05-29T07:01:38Z","title":"Welfare, Improvability, and Variance: A Principal-Agent Approach to Optimal Benchmark Item Aggregation","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-06-28T23:49:57.580051Z"},"links":{"cited_paper":"/paper/2502.06559","citing_paper":"/paper/2605.30916"},"observation_digest":"sha256:7c3a693377f9eed0c1027357c95e2956bd2beb058f7199d7f0fa4e1c2e373253","observation_id":"4b7af485-4a67-4466-9444-86cac6b58ff3","resolution":{"observed_at":"2026-06-28T23:52:49.468378Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-04T01:08:05.510513+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-04T01:08:05.510513+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"cited_work":{"arxiv_id":"2502.06559","doi":"10.48550/arxiv.2502.06559","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.06559","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Can we trust ai benchmarks? an interdisciplinary review of current issues in ai evaluation","venue":"ArXiv.org","work_id":"cdcf1d27-8950-4301-b2c5-4fa9aad92cab","year":2025},"citing_paper":{"arxiv_id":"2606.07996","last_updated":"2026-06-06T06:27:54Z","snapshot_observed_at":"2026-08-12T17:53:51.469237Z","submitted_at":"2026-06-06T06:27:54Z","title":"MC-PDD: Masked Corpus-Level Pretraining Data Detection for Black-Box Large Language Models","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-06-27T20:02:50.169589Z"},"links":{"cited_paper":"/paper/2502.06559","citing_paper":"/paper/2606.07996"},"observation_digest":"sha256:1d8c173970647e44722707f43714437c57c67d9c39a4da8bfdc2684c2438c9e6","observation_id":"34ae6e45-61a6-4223-b38c-fccf5d80e4e5","resolution":{"observed_at":"2026-07-02T20:57:23.308800Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-04T01:08:05.510513+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-04T01:08:05.510513+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"cited_work":{"arxiv_id":"2502.06559","doi":"10.48550/arxiv.2502.06559","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.06559","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Can we trust ai benchmarks? an interdisciplinary review of current issues in ai evaluation","venue":"ArXiv.org","work_id":"cdcf1d27-8950-4301-b2c5-4fa9aad92cab","year":2025},"citing_paper":{"arxiv_id":"2606.09118","last_updated":"2026-07-04T02:42:19Z","snapshot_observed_at":"2026-08-15T16:38:25.826951Z","submitted_at":"2026-06-08T07:11:56Z","title":"ComplexConstraints and Beyond: Expert Rubrics for RLVR","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-27T16:37:11.141846Z"},"links":{"cited_paper":"/paper/2502.06559","citing_paper":"/paper/2606.09118"},"observation_digest":"sha256:fd2984ede3c97bcf96024c8f363cf5e4cad9a813c4d946c233ef2fa19e38640f","observation_id":"d4ea5a4e-f8cb-4247-a929-6a3ea6b00793","resolution":{"observed_at":"2026-07-03T01:17:31.360048Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-04T01:08:05.510513+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-04T01:08:05.510513+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"cited_work":{"arxiv_id":"2502.06559","doi":"10.48550/arxiv.2502.06559","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.06559","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Can we trust ai benchmarks? an interdisciplinary review of current issues in ai evaluation","venue":"ArXiv.org","work_id":"cdcf1d27-8950-4301-b2c5-4fa9aad92cab","year":2025},"citing_paper":{"arxiv_id":"2606.18158","last_updated":"2026-06-16T16:57:12Z","snapshot_observed_at":"2026-08-13T08:27:57.453143Z","submitted_at":"2026-06-16T16:57:12Z","title":"The Measurement Gap in the Automation of EU Law: Benchmarking Doctrinal Legal Reasoning under the EU AI Act","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-06-26T22:16:25.919667Z"},"links":{"cited_paper":"/paper/2502.06559","citing_paper":"/paper/2606.18158"},"observation_digest":"sha256:296d6d302904ec76c2306b4c2de04c5dd639cc492249e13ab4bd5fae0694ab2c","observation_id":"121011bc-ae69-4a14-a5cb-0acd8e8475b0","resolution":{"observed_at":"2026-07-03T23:29:02.483341Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-04T01:08:05.510513+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-04T01:08:05.510513+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"cited_work":{"arxiv_id":"2502.06559","doi":"10.48550/arxiv.2502.06559","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.06559","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Can we trust ai benchmarks? an interdisciplinary review of current issues in ai evaluation","venue":"ArXiv.org","work_id":"cdcf1d27-8950-4301-b2c5-4fa9aad92cab","year":2025},"citing_paper":{"arxiv_id":"2606.31755","last_updated":"2026-06-30T14:44:56Z","snapshot_observed_at":"2026-08-13T02:54:58.584666Z","submitted_at":"2026-06-30T14:44:56Z","title":"A Technical Typology of AI Systems in Public Administration","version":1},"reference_index":282,"source":"arxiv_source","source_observed_at":"2026-07-01T02:44:10.129474Z"},"links":{"cited_paper":"/paper/2502.06559","citing_paper":"/paper/2606.31755"},"observation_digest":"sha256:2b5a1854dac1563e6a43b3aebdc2f75754244c2f09d1fc1ca366b07a6adb9667","observation_id":"6390513f-e9d5-4895-89ed-b04f8dfc184f","resolution":{"observed_at":"2026-07-01T02:45:17.178772Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-04T01:08:05.510513+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-04T01:08:05.510513+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.06559","snapshot_observed_at":"2026-07-12T05:51:08.101805Z","title":"8 The Foreign Policy AI Evaluation Gap Fearon, J","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.02955","last_updated":"2026-07-03T04:59:03Z","snapshot_observed_at":"2026-08-06T03:14:27.339658Z","submitted_at":"2026-07-03T04:59:03Z","title":"The Foreign Policy AI Evaluation Gap","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-07-12T05:51:08.101805Z"},"links":{"cited_paper":"/paper/2502.06559","citing_paper":"/paper/2607.02955"},"observation_digest":"sha256:6fa9166e7a72d18dabc446420f83dcf545b305a126a2792eb6adf86e72807165","observation_id":"3ff06c53-3f47-4ec8-9924-1a675112e4dc","resolution":{"observed_at":"2026-07-12T05:51:08.101805Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.06559","snapshot_observed_at":"2026-08-01T20:23:02.816523Z","title":"arXiv preprint arXiv:2502.06559 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.16652","last_updated":"2026-07-18T06:09:07Z","snapshot_observed_at":"2026-08-15T05:49:53.208736Z","submitted_at":"2026-07-18T06:09:07Z","title":"Position: Explanation Stability Is a Property of the Model Method Pair, Not the Model","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-01T20:23:02.816523Z"},"links":{"cited_paper":"/paper/2502.06559","citing_paper":"/paper/2607.16652"},"observation_digest":"sha256:5ed9132330c671258474b18e8752b56addc9d5aa66358d0c5627b6f3e1983d78","observation_id":"d68874ed-8aa6-40c8-aee3-a439b3174a5d","resolution":{"observed_at":"2026-08-01T20:23:02.816523Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.06559","snapshot_observed_at":"2026-08-04T00:44:37.517694Z","title":", author Purificato, E","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00680","last_updated":"2026-08-01T14:01:17Z","snapshot_observed_at":"2026-08-16T20:49:17.659419Z","submitted_at":"2026-08-01T14:01:17Z","title":"Multi-Dimensional Assessment for AI Cognition (MAAC): A Theoretical Framework for Process-Oriented Cognitive Evaluation of Text-Based AI Systems","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-04T00:44:37.517694Z"},"links":{"cited_paper":"/paper/2502.06559","citing_paper":"/paper/2608.00680"},"observation_digest":"sha256:369f1226d69c33133cff2c1536b4be0e4cdea73b6bace9bdfe6c15a9ca2cefc1","observation_id":"3d59bbba-dbce-43eb-b893-e9613d902022","resolution":{"observed_at":"2026-08-04T00:44:37.517694Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.06559","snapshot_observed_at":"2026-08-04T00:44:45.631232Z","title":"arXiv preprint arXiv:2502.06559 , year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00680","last_updated":"2026-08-01T14:01:17Z","snapshot_observed_at":"2026-08-16T20:49:17.659419Z","submitted_at":"2026-08-01T14:01:17Z","title":"Multi-Dimensional Assessment for AI Cognition (MAAC): A Theoretical Framework for Process-Oriented Cognitive Evaluation of Text-Based AI Systems","version":1},"reference_index":102,"source":"arxiv_source","source_observed_at":"2026-08-04T00:44:45.631232Z"},"links":{"cited_paper":"/paper/2502.06559","citing_paper":"/paper/2608.00680"},"observation_digest":"sha256:00de87ab33d2c9781da6141cc8c576c149ac085bd4c139ca5b0fee42b5152c7b","observation_id":"793a899a-75c2-4ab1-b017-0828169d0495","resolution":{"observed_at":"2026-08-04T00:44:45.631232Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2502.06559/citation-record","integrity":"/paper/2502.06559/integrity","json":"/paper/2502.06559/citation-record.json","paper":"/paper/2502.06559"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1126/science.ade2420","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"Thompson","venue":"Science","work_id":"74831883-9f17-42c7-8d4d-763620c1be2b","year":2023},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.711009Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:3a2846765eeee3ab368d2612da20e38b34202c22b4b1fd1254b124108cbea266","observation_id":"0c37d026-2980-4063-a91d-9dbb3bb3df9a","resolution":{"observed_at":"2026-08-08T15:06:55.827575Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:54.717113Z","title":"Field-building and the epistemic culture of AI safety","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.717113Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:b433639b6fad1f005cbfb1ee10e585692293a4f50a91042cde4b26077f1c294a","observation_id":"484125d1-2ae7-4df3-b436-b1d77ad2717d","resolution":{"observed_at":"2026-08-08T15:06:54.717113Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01781","last_updated":"2024-07-03T11:20:43Z","snapshot_observed_at":"2026-08-18T18:45:01.785556Z","submitted_at":"2024-02-01T19:12:25Z","title":"When Benchmarks are Targets: Revealing the Sensitivity of Large Language Model Leaderboards","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01781","snapshot_observed_at":"2026-08-08T15:06:54.722646Z","title":"Saiful Bari, and Haidar Khan","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.722646Z"},"links":{"cited_paper":"/paper/2402.01781","citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:8f500d60f5bea1224eb3f7bfdf516136636032ec0d723e16ab5922cebe54db36","observation_id":"6c1b2e4c-07fe-4278-963b-01609ebcdb51","resolution":{"observed_at":"2026-08-08T15:06:54.722646Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1093/nar/gkq625","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":null,"venue":"Nucleic Acids Research","work_id":"d380ea71-21f7-4c36-b0e8-4f0169a6cb0e","year":2010},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.728474Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:cacaad740d2db47214b5fc5ea915cb9ad09c2280aad999cd492999e0ecf3d809","observation_id":"696c2d53-8a8d-462a-9d7b-1ce5de68e947","resolution":{"observed_at":"2026-08-08T15:06:55.799348Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:54.733654Z","title":"Truth Is a Lie : Crowd Truth and the Seven Myths of Human Annotation","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.733654Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:ba31d36d44d15c40d78468998304d6dce0b2c4440139e7ece8b8f1b8475a7b95","observation_id":"c8b2ce17-cafa-4bcf-bf17-9e10d658e37d","resolution":{"observed_at":"2026-08-08T15:06:54.733654Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.05224","last_updated":"2024-11-07T22:36:19Z","snapshot_observed_at":"2026-08-19T17:17:22.008958Z","submitted_at":"2024-11-07T22:36:19Z","title":"Beyond the Numbers: Transparency in Relation Extraction Benchmark Creation and Leaderboards","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.05224","snapshot_observed_at":"2026-08-08T15:06:54.739088Z","title":"Beyond the Numbers : Transparency in Relation Extraction Benchmark Creation and Leaderboards , November 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.739088Z"},"links":{"cited_paper":"/paper/2411.05224","citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:3893f2e3f7cfeeaa821e4ae7e64c922f245f4414736298bcb8f08718d029a6bb","observation_id":"a20338bb-4173-4f0b-91bf-111af29a152d","resolution":{"observed_at":"2026-08-08T15:06:54.739088Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:54.744859Z","title":"Experiences from using snowballing and database searches in systematic literature studies","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.744859Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:455c60e2758e9fd90fba55b26b2ab1067edc498252bd448ef9f7c8704d7cec58","observation_id":"9809ec0c-2c0c-4853-bdb8-0cdf0367ef5d","resolution":{"observed_at":"2026-08-08T15:06:54.744859Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.05498","last_updated":"2022-04-28T18:04:20Z","snapshot_observed_at":"2026-08-16T18:18:12.221348Z","submitted_at":"2021-06-10T04:59:06Z","title":"It's COMPASlicated: The Messy Relationship between RAI Datasets and Algorithmic Fairness Benchmarks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.05498","snapshot_observed_at":"2026-08-08T15:06:54.749660Z","title":"It's COMPASlicated : The Messy Relationship between RAI Datasets and Algorithmic Fairness Benchmarks , April 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.749660Z"},"links":{"cited_paper":"/paper/2106.05498","citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:e21bed71f20c92f935c59f8615f99e74b57f3aac0d13143c53327bb14e05f704","observation_id":"daa8d07b-660c-4e6a-b008-0f38dba07b34","resolution":{"observed_at":"2026-08-08T15:06:54.749660Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2007.03488","last_updated":"2020-12-16T22:36:27Z","snapshot_observed_at":"2026-08-21T06:15:19.754254Z","submitted_at":"2020-07-07T14:20:26Z","title":"Benchmarking in Optimization: Best Practice and Open Issues","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2007.03488","snapshot_observed_at":"2026-08-08T15:06:54.755116Z","title":"Malan, Jason H","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.755116Z"},"links":{"cited_paper":"/paper/2007.03488","citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:c40912eb1c7b0bee236b353e1a129d55ef41393a5dbd5e6d0866cf30e88bd59b","observation_id":"6a11a392-f6f5-4e89-8127-98f453d3d61e","resolution":{"observed_at":"2026-08-08T15:06:54.755116Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:54.760557Z","title":"The Death of the Static AI Benchmark , March 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.760557Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:435d8729c8b73d57caf773b5127d3b2c521c8c72f59434cb0dc751574bb99a65","observation_id":"bb336076-1ff3-44af-9a59-a08018a608af","resolution":{"observed_at":"2026-08-08T15:06:54.760557Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.14782","last_updated":"2026-05-31T00:04:33Z","snapshot_observed_at":"2026-08-16T13:50:06.099874Z","submitted_at":"2024-05-23T16:50:49Z","title":"Lessons from the Trenches on Reproducible Evaluation of Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.14782","snapshot_observed_at":"2026-08-08T15:06:54.765506Z","title":"Lee, Haonan Li, Charles Lovering, Niklas Muennighoff, Ellie Pavlick, Jason Phang, Aviya Skowron, Samson Tan, Xiangru Tang, Kevin A","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.765506Z"},"links":{"cited_paper":"/paper/2405.14782","citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:5393e259cdfb8ea6d64c4b06f609f2309677d1cc48724045c55434325f2fbc88","observation_id":"80194e49-4368-4eca-9465-d551e6fe5822","resolution":{"observed_at":"2026-08-08T15:06:54.765506Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.14462","last_updated":"2024-01-25T19:00:29Z","snapshot_observed_at":"2026-08-20T10:18:09.101803Z","submitted_at":"2024-01-25T19:00:29Z","title":"AI auditing: The Broken Bus on the Road to AI Accountability","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.14462","snapshot_observed_at":"2026-08-08T15:06:54.770662Z","title":"AI auditing: The Broken Bus on the Road to AI Accountability , January 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.770662Z"},"links":{"cited_paper":"/paper/2401.14462","citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:7ea114e14cd1a7aa3223abc76c1b39225764af992adab7cf199b9acd4be0e827","observation_id":"a0890d5f-d21e-442e-8ac1-12a9e451f0ac","resolution":{"observed_at":"2026-08-08T15:06:54.770662Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:54.775599Z","title":"Benchmark datasets driving artificial intelligence development fail to capture the needs of medical professionals","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.775599Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:e13f4d5febcf1e4aa0bf8f7cfc967eaee0413f09f1bed32b42800667b4ba3481","observation_id":"c3e25ed8-e2f8-4f44-b8d4-c27049d49e12","resolution":{"observed_at":"2026-08-08T15:06:54.775599Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:54.780076Z","title":"Making Intelligence : Ethical Values in IQ and ML Benchmarks","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.780076Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:a8d218780673192c0990f3cf5d2032de2e16db3732d8ca932e8126c8c30ea977","observation_id":"7f686d32-1905-479c-84be-66c6b6fb57bd","resolution":{"observed_at":"2026-08-08T15:06:54.780076Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:54.784923Z","title":"Stereotyping Norwegian Salmon : An Inventory of Pitfalls in Fairness Benchmark Datasets","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.784923Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:6202771b48271811926d46545ef465a6efdeeac6e6508e8a4278957751f73efa","observation_id":"97b5eb97-07a8-4e52-93ca-cd48646f8309","resolution":{"observed_at":"2026-08-08T15:06:54.784923Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:54.789694Z","title":"Bowman and George Dahl","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.789694Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:c153966e6f7f1b9f627aa45a0f14cde8f0eca17514aaea3ed1e825a5d39640ec","observation_id":"7e708afa-7fdb-4a90-9a59-1a46fa1da1d1","resolution":{"observed_at":"2026-08-08T15:06:54.789694Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1007/978-94-007-0753-5_170","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"Benchmarking, pages 363--368","venue":null,"work_id":"a49b666f-242e-41b3-9da3-ce01d0fcdbc8","year":2014},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.794743Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:4b0f4705dee622cfdad57c01c153924c32bdd79ae9b50ca74105000fc03a765f","observation_id":"3accfd96-c13e-4391-b0c0-8cbb94169648","resolution":{"observed_at":"2026-08-08T15:06:55.750403Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.09221","last_updated":"2024-07-12T12:37:13Z","snapshot_observed_at":"2026-08-20T11:55:51.660918Z","submitted_at":"2024-07-12T12:37:13Z","title":"Evaluating AI Evaluation: Perils and Prospects","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.09221","snapshot_observed_at":"2026-08-08T15:06:54.799539Z","title":"Evaluating AI Evaluation : Perils and Prospects , July 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.799539Z"},"links":{"cited_paper":"/paper/2407.09221","citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:24543ca763bb93228cea9435e00baab3787bb200df7f596021b87d5ccbfe2b53","observation_id":"4ac40191-ef53-4541-a50d-b0b9a7814651","resolution":{"observed_at":"2026-08-08T15:06:54.799539Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:54.804900Z","title":"Ullman, Fernando Martinez-Plumed, Joshua B","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.804900Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:c872d40cdc4292ade7096bc31dccaea2740066ca783c2523105e3b44cdc4027c","observation_id":"565f89de-82ff-4744-8bc2-0a111fc168ee","resolution":{"observed_at":"2026-08-08T15:06:54.804900Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:54.809515Z","title":null,"venue":null,"work_id":null,"year":1989},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.809515Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:d44297b756a8588a544c047d6731710c0bf566286af383f85e774749da950499","observation_id":"9c3289a9-82a7-4ffc-ba2a-02466e0b23d9","resolution":{"observed_at":"2026-08-08T15:06:54.809515Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.03109","last_updated":"2023-12-29T02:12:03Z","snapshot_observed_at":"2026-08-20T09:21:16.017380Z","submitted_at":"2023-07-06T16:28:35Z","title":"A Survey on Evaluation of Large Language Models","version":9},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.03109","snapshot_observed_at":"2026-08-08T15:06:54.819434Z","title":"Yu, Qiang Yang, and Xing Xie","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.819434Z"},"links":{"cited_paper":"/paper/2307.03109","citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:4285900ee1ddc41b0e1da2ee82a2202ff082681ab2a4d692948341a86ef80efb","observation_id":"dc8740b4-a76e-4116-ba6a-bb63733cbcbc","resolution":{"observed_at":"2026-08-08T15:06:54.819434Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1038/s41928-022-00798-8","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"Cheng, CS","venue":"Nature Electronics","work_id":"2df6a8f2-fc20-4e76-abda-59182b94b709","year":2020},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.824738Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:721151e297d229f2fcca8e43db0b556c9f0303fb9d4f3d5cf3027401db6557fa","observation_id":"25fb02ee-dad9-4db2-856e-9344e71bd96e","resolution":{"observed_at":"2026-08-08T15:06:55.721117Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[{"edge_observation":{"observed_at":"2026-08-08T18:17:49.878181+00:00","source":"paper_reference_links","state":"open"},"event_date":"2022-08-24","event_type":"correction","notice_doi":"10.1038/s41928-022-00839-2","provenance":{"observed_at":"2026-07-11T03:08:33.550039+00:00","source":"crossref","source_record_id":"10.1038/s41928-022-00839-2->10.1038/s41928-022-00798-8:correction"}}],"reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1911.01547","last_updated":"2019-11-25T13:02:04Z","snapshot_observed_at":"2026-08-17T05:38:14.090895Z","submitted_at":"2019-11-05T00:31:38Z","title":"On the Measure of Intelligence","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1911.01547","snapshot_observed_at":"2026-08-08T15:06:54.829633Z","title":"On the Measure of Intelligence , November 2019","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.829633Z"},"links":{"cited_paper":"/paper/1911.01547","citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:659772c2257572ce76240d9ddc0d90f2599989d4b9ee117227460383a2b02615","observation_id":"1306bf06-d65b-4073-887c-e60cfb9a90b6","resolution":{"observed_at":"2026-08-08T15:06:54.829633Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1017/s1351324919000275","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"A survey of 25 years of evaluation","venue":"Natural Language Engineering","work_id":"1cb419b2-fa79-4d4d-9978-c837a1a1ddac","year":2019},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.836297Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:b0b58746056c20f21903504bcdebc53708baac1b165d7512d275cf2e56d59437","observation_id":"0a2d6b0f-b196-4539-aadc-2ef477525f55","resolution":{"observed_at":"2026-08-08T15:06:55.704740Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2107.07002","last_updated":"2021-07-14T21:08:30Z","snapshot_observed_at":"2026-08-19T06:03:25.160551Z","submitted_at":"2021-07-14T21:08:30Z","title":"The Benchmark Lottery","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2107.07002","snapshot_observed_at":"2026-08-08T15:06:54.842393Z","title":"Gritsenko, Zhe Zhao, Neil Houlsby, Fernando Diaz, Donald Metzler, and Oriol Vinyals","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.842393Z"},"links":{"cited_paper":"/paper/2107.07002","citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:4b9f7523995e763c54a343b3eee693dc6fdd76eb3b4c915d6100794a9fd521de","observation_id":"06c69184-38f8-4c73-b390-76531cac405f","resolution":{"observed_at":"2026-08-08T15:06:54.842393Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:54.847430Z","title":"On the genealogy of machine learning datasets: A critical history of ImageNet","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.847430Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:80afc56189ed153f5aebe4c79bb58d067947d51f150a5272db2dee074a179b91","observation_id":"7472ef13-715b-4005-84b2-d59542854a89","resolution":{"observed_at":"2026-08-08T15:06:54.847430Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2007.07399","last_updated":"2020-07-14T23:22:13Z","snapshot_observed_at":"2026-08-20T10:20:54.991015Z","submitted_at":"2020-07-14T23:22:13Z","title":"Bringing the People Back In: Contesting Benchmark Machine Learning Datasets","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2007.07399","snapshot_observed_at":"2026-08-08T15:06:54.852188Z","title":"Bringing the People Back In : Contesting Benchmark Machine Learning Datasets , July 2020","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.852188Z"},"links":{"cited_paper":"/paper/2007.07399","citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:6a7c167125c371e6be9feaefd4e9ac0dacc38085d14bece899238cf17b78b5e7","observation_id":"bbf73567-e96b-45ca-91cc-b9c5e206b3b1","resolution":{"observed_at":"2026-08-08T15:06:54.852188Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1177/20539517241242457","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"Agreements ‘in the wild’: Standards and alignment in machine learning benchmark dataset construction","venue":"Big Data & Society","work_id":"9a6ce117-1e02-4463-b606-aef97359cbf3","year":2024},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.861689Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:fc1224ae214014978d635b1019fe7576f22e2a22cf3594750888e5ac65f8ce07","observation_id":"c76f87d3-f4c7-450c-9d02-f550463139cc","resolution":{"observed_at":"2026-08-08T15:06:55.677229Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2009.13888","last_updated":"2021-03-03T06:22:32Z","snapshot_observed_at":"2026-08-21T14:21:24.670599Z","submitted_at":"2020-09-29T09:25:31Z","title":"Utility is in the Eye of the User: A Critique of NLP Leaderboards","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2009.13888","snapshot_observed_at":"2026-08-08T15:06:54.866504Z","title":"Utility is in the Eye of the User : A Critique of NLP Leaderboards , March 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.866504Z"},"links":{"cited_paper":"/paper/2009.13888","citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:a9aace184b87520cd38c06aecf55b9dfe9e4c847a577d700d30b9ab47273f8c1","observation_id":"0017ebb3-a924-48e8-80d4-3ef1c890c58b","resolution":{"observed_at":"2026-08-08T15:06:54.866504Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:54.871752Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.871752Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:13d6af7c2b664b894f63ca630d014330527bc793c34e36324e7e3ccf631db900","observation_id":"e99ca6c3-cee7-472b-9441-7c5e05461bba","resolution":{"observed_at":"2026-08-08T15:06:54.871752Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:54.876695Z","title":"First Draft of the General-Purpose AI Code of Practice published, written by independent experts , 2024 a","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.876695Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:8eda864253563babdb0aa36b7f0e76549040b5282a1ff446b37949320348bb30","observation_id":"8c2ccb33-28c5-487f-b0b4-149dcab46d89","resolution":{"observed_at":"2026-08-08T15:06:54.876695Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:54.881300Z","title":"Second Draft of the General-Purpose AI Code of Practice published, written by independent experts , 2024 b","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.881300Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:97e50a4ae4dd48300ffd606a0cfb8beb07caeed37b274df7c6570c1074128b11","observation_id":"58203012-b7fd-4191-9b28-d2b0b6967556","resolution":{"observed_at":"2026-08-08T15:06:54.881300Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:54.885976Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.885976Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:d3098d8e5d269fd092c85fea2823c0d2d06794f7eec45865aacbe251e5998b12","observation_id":"16135675-89cf-4cb4-af1a-1b4ab4decb6d","resolution":{"observed_at":"2026-08-08T15:06:54.885976Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:54.890605Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.890605Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:a2a504924b56c04eded1c23ab42c406f46caf5d476eae167054372ca70b223fe","observation_id":"5e8c7efa-28f3-4950-8a14-77141566b3a6","resolution":{"observed_at":"2026-08-08T15:06:54.890605Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1803.09010","last_updated":"2021-12-01T20:29:42Z","snapshot_observed_at":"2026-08-15T04:24:26.471843Z","submitted_at":"2018-03-23T23:22:18Z","title":"Datasheets for Datasets","version":8},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1803.09010","snapshot_observed_at":"2026-08-08T15:06:54.895262Z","title":"Datasheets for Datasets , December 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.895262Z"},"links":{"cited_paper":"/paper/1803.09010","citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:d1c6fa6d36ad7e7cdd8f018ae70dfe994b8a9bd5ca37f3d262d25ca0107ce704","observation_id":"68abdec4-5a97-4cbb-a2c0-d07552426a14","resolution":{"observed_at":"2026-08-08T15:06:54.895262Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:54.900368Z","title":"Repairing the Cracked Foundation : A Survey of Obstacles in Evaluation Practices for Generated Text","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.900368Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:4ad8d975d7ce7c72a85ac32e8b4e66ec173dc3fb6c51995f7e8871b0645409c0","observation_id":"3de85b09-ff27-4466-a91e-f2fd5e4525cd","resolution":{"observed_at":"2026-08-08T15:06:54.900368Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2004.07780","last_updated":"2023-11-21T15:22:43Z","snapshot_observed_at":"2026-08-17T23:50:29.006196Z","submitted_at":"2020-04-16T17:18:49Z","title":"Shortcut Learning in Deep Neural Networks","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2004.07780","snapshot_observed_at":"2026-08-08T15:06:54.905352Z","title":"Wichmann","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.905352Z"},"links":{"cited_paper":"/paper/2004.07780","citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:e6a41710b5c9c589042bbf091530a05f4e163ac44816210b56b483471f37519f","observation_id":"feab772e-8ce3-47b6-9c30-53d848da43fc","resolution":{"observed_at":"2026-08-08T15:06:54.905352Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04127","last_updated":"2025-01-10T14:31:21Z","snapshot_observed_at":"2026-08-18T22:14:47.638195Z","submitted_at":"2024-06-06T14:49:06Z","title":"Are We Done with MMLU?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04127","snapshot_observed_at":"2026-08-08T15:06:54.910434Z","title":"Are We Done with MMLU ?, June 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.910434Z"},"links":{"cited_paper":"/paper/2406.04127","citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:90628f3b71a68ff800a51382d4b3a3a3ea6387620737d3e3602dbe13b124a636","observation_id":"bc15d71d-32bd-4810-b3cf-8efe834a4e33","resolution":{"observed_at":"2026-08-08T15:06:54.910434Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.2760/796551","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"Diversity in artificial intelligence conferences","venue":"RePEc: Research Papers in Economics","work_id":"221d016b-0d93-4727-9069-60b398a0b796","year":2024},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.916120Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:227a7a689ce656980ebc51710373a4cd149cdd405c8e8ace6c3853f392a65b8d","observation_id":"cd55a78b-da1c-4179-9bdc-c9f90c07035b","resolution":{"observed_at":"2026-08-08T15:06:55.649867Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.14093","last_updated":"2024-12-20T02:22:19Z","snapshot_observed_at":"2026-08-13T11:12:58.507759Z","submitted_at":"2024-12-18T17:41:24Z","title":"Alignment faking in large language models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.14093","snapshot_observed_at":"2026-08-08T15:06:54.921165Z","title":"Bowman, and Evan Hubinger","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.921165Z"},"links":{"cited_paper":"/paper/2412.14093","citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:c9aeb782ff762c46341e4aec615306eebdb1d4a9b23b41d87c29f5743e8aade4","observation_id":"9ba3440b-6225-41f8-a049-d9647f762fcf","resolution":{"observed_at":"2026-08-08T15:06:54.921165Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07959","last_updated":"2025-02-03T14:51:44Z","snapshot_observed_at":"2026-08-17T22:58:18.976534Z","submitted_at":"2024-10-10T14:23:51Z","title":"COMPL-AI Framework: A Technical Interpretation and LLM Benchmarking Suite for the EU Artificial Intelligence Act","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.07959","snapshot_observed_at":"2026-08-08T15:06:54.930734Z","title":"COMPL - AI Framework : A Technical Interpretation and LLM Benchmarking Suite for the EU Artificial Intelligence Act , October 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.930734Z"},"links":{"cited_paper":"/paper/2410.07959","citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:ee914c87b6a91698eb05a165995b99311cfb0fbdd2bb8ad3cc1709bb4243b098","observation_id":"0efd86a3-2cc2-4eef-ab17-b0890489515e","resolution":{"observed_at":"2026-08-08T15:06:54.930734Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:54.935965Z","title":"Ai hype is built on high test scores","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.935965Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:807b2f3fa1bd6261eaef1b009d20f7c6f297666ac9d3c0887d834a16c975c368","observation_id":"9ae31a2e-06bf-4a6d-8d4a-841d6cc8b96f","resolution":{"observed_at":"2026-08-08T15:06:54.935965Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2009.03300","last_updated":"2021-01-12T18:57:11Z","snapshot_observed_at":"2026-08-13T20:44:28.824685Z","submitted_at":"2020-09-07T17:59:25Z","title":"Measuring Massive Multitask Language Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2009.03300","snapshot_observed_at":"2026-08-08T15:06:54.941201Z","title":"Measuring Massive Multitask Language Understanding , January 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.941201Z"},"links":{"cited_paper":"/paper/2009.03300","citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:50e2c3eb8f238ec4838ca447081f89868721e2682bc7350964ec4c8c773dc247","observation_id":"5d51f0fb-7731-40cb-a0dd-ddf5555d5eed","resolution":{"observed_at":"2026-08-08T15:06:54.941201Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:54.946419Z","title":null,"venue":null,"work_id":null,"year":2000},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.946419Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:09fc4751085c9ebafa94d2f59d3130b2dd1acb90d0b92af5c2bdf267423aa1bf","observation_id":"d81ad73e-70a8-4b69-bf92-e3653a813d0a","resolution":{"observed_at":"2026-08-08T15:06:54.946419Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.05694","last_updated":"2024-07-30T02:37:56Z","snapshot_observed_at":"2026-08-16T13:36:16.801244Z","submitted_at":"2024-07-08T07:53:06Z","title":"On the Limitations of Compute Thresholds as a Governance Strategy","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.05694","snapshot_observed_at":"2026-08-08T15:06:54.951473Z","title":"On the Limitations of Compute Thresholds as a Governance Strategy , July 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.951473Z"},"links":{"cited_paper":"/paper/2407.05694","citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:907804310ca43f8dc1988a6eeabd959c0bfdf3cb4e33eb2379e75368ed96d0cf","observation_id":"0219469c-622c-4529-b07d-cefb2c727659","resolution":{"observed_at":"2026-08-08T15:06:54.951473Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:54.957293Z","title":"Evaluation Gaps in Machine Learning Practice","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.957293Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:6e50d8f54518c07b5f627c72fef865ee710420d9e88bc42e7c95e6ba6eaa9197","observation_id":"2db1e91e-0e60-4dac-bea1-0c5291fd1b00","resolution":{"observed_at":"2026-08-08T15:06:54.957293Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2251.23722","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:57.767298Z","title":"Systematic literature studies: database searches vs","venue":null,"work_id":"fe6b34a0-48da-40ea-a027-02d86f6c3866","year":2012},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.962237Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:46f6851ece6014eec6993d0df908b8f703f1b9519ec0c1d6c296ff3afb82b865","observation_id":"e4dca48b-bf4e-40c1-bfd1-d618aafcf675","resolution":{"observed_at":"2026-08-08T15:06:57.775230Z","resolver_source":"raw_fallback","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:54.967056Z","title":"Escaping the McNamara Fallacy : Toward More Impactful Recommender Systems Research","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.967056Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:c02236a446cdebf8312c48880ddbdb2cbbe2250e15030a557d12fc08c311cf42","observation_id":"dacddb1d-b479-49e0-8b32-7e5bcece080f","resolution":{"observed_at":"2026-08-08T15:06:54.967056Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:54.971913Z","title":"The Constitution of Algorithms : Ground - Truthing , Programming , Formulating","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.971913Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:dd512483f705b1f06830a975fc2baf290657b36b8ae076efda26bdeda37d8540","observation_id":"e9c62eaa-2b20-4415-a428-c74b139ceab0","resolution":{"observed_at":"2026-08-08T15:06:54.971913Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:54.976493Z","title":"Under the radar? examining the evaluation of foundation models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.976493Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:efee4e00e51b0cac245f80f56ab611be8b2b2eaa580eb1e56e02088ee2d20740","observation_id":"750eb3c3-0ff1-4a11-bc02-9ee57655392f","resolution":{"observed_at":"2026-08-08T15:06:54.976493Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1177/20539517221146122","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"Ground truth tracings ( GTT ): On the epistemic limits of machine learning","venue":"Big Data & Society","work_id":"e81ded18-8e1c-4057-8b5d-05d666868e4c","year":2023},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.981043Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:36515ab03ef8173cc8b76c5d3a7f90d68e240e8e9a00627a16fb1cd4e7f909cd","observation_id":"b0477496-d9a3-4df8-8ce3-56e75c20fccb","resolution":{"observed_at":"2026-08-08T15:06:55.613881Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.01502","last_updated":"2024-07-01T17:48:14Z","snapshot_observed_at":"2026-08-19T14:24:18.304488Z","submitted_at":"2024-07-01T17:48:14Z","title":"AI Agents That Matter","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.01502","snapshot_observed_at":"2026-08-08T15:06:54.985509Z","title":"Siegel, Nitya Nadgir, and Arvind Narayanan","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.985509Z"},"links":{"cited_paper":"/paper/2407.01502","citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:757288f3a4339236e645dc49f1c88438a322826a8b00bd0ab827e319087338a5","observation_id":"672fc10c-0bb0-4269-b1d3-0aab8d3a844c","resolution":{"observed_at":"2026-08-08T15:06:54.985509Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:54.990287Z","title":"Leakage in data mining: Formulation, detection, and avoidance","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.990287Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:d776612fca26c3d62c9f91fc8256b489e3cea8cf688b03c0a116dc324ae0f786","observation_id":"ff0e720a-aff5-4ce8-a171-f858ce4d4746","resolution":{"observed_at":"2026-08-08T15:06:54.990287Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:54.995109Z","title":"Everyone Is Judging AI by These Tests","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.995109Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:b85c9861cebb6368da034b356f2d93d4e138b78edd63abb413ec49fc028e5798","observation_id":"b1a8f690-3432-4c7c-8e35-71513d7a23af","resolution":{"observed_at":"2026-08-08T15:06:54.995109Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1038/s41598-024-58937-4","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"Mulvehill, and Deborah L","venue":"Scientific Reports","work_id":"e9d972e5-1311-4310-94a1-4f6583e251af","year":2024},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:54.999946Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:aa1ab4020dd13a441058832d438db3b33136cbc39613101edd90d246b100e9fa","observation_id":"cf245071-b30c-481b-a6aa-c0c0d7a5245c","resolution":{"observed_at":"2026-08-08T15:06:55.598819Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1177/20539517221113772","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"Feeling fixes: Mess and emotion in algorithmic audits","venue":"Big Data & Society","work_id":"13e21ff2-e062-4c6a-97ed-99a5ac87add2","year":2022},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.005003Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:c232ce469206c744e535bdce0e2e55c32a213f366c78e3e4ec692dca34fa2a2d","observation_id":"a6a2a399-860f-4bfa-83e6-33a1cddae108","resolution":{"observed_at":"2026-08-08T15:06:55.582988Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.01716","last_updated":"2021-12-03T05:01:47Z","snapshot_observed_at":"2026-08-20T03:26:23.790675Z","submitted_at":"2021-12-03T05:01:47Z","title":"Reduced, Reused and Recycled: The Life of a Dataset in Machine Learning Research","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.01716","snapshot_observed_at":"2026-08-08T15:06:55.009989Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":60,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.009989Z"},"links":{"cited_paper":"/paper/2112.01716","citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:2c0b1425824f57ffa5bf343e40b029a7f56beebc81c5d09eccb475311db832dc","observation_id":"5e517da3-25ce-4e17-800d-f7f0a2ebb021","resolution":{"observed_at":"2026-08-08T15:06:55.009989Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.06647","last_updated":"2024-04-11T02:09:23Z","snapshot_observed_at":"2026-08-19T22:44:59.606833Z","submitted_at":"2024-04-09T22:55:06Z","title":"From Protoscience to Epistemic Monoculture: How Benchmarking Set the Stage for the Deep Learning Revolution","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.06647","snapshot_observed_at":"2026-08-08T15:06:55.015239Z","title":"Koch and David Peterson","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":61,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.015239Z"},"links":{"cited_paper":"/paper/2404.06647","citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:24d03a089d0a4cf929b9038478bfaca5402f75f4264a3ffa70b89cc9f008031c","observation_id":"2e1c6beb-794c-4e4e-add0-2cead3d7f5c8","resolution":{"observed_at":"2026-08-08T15:06:55.015239Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2204.05151","last_updated":"2022-04-11T14:36:39Z","snapshot_observed_at":"2026-08-16T17:07:46.488739Z","submitted_at":"2022-04-11T14:36:39Z","title":"Metaethical Perspectives on 'Benchmarking' AI Ethics","version":1},"cited_work":{"arxiv_id":"2204.05151","doi":null,"metadata_source":"pith","pith_arxiv_id":"2204.05151","snapshot_observed_at":"2026-08-08T15:06:57.555285Z","title":"Metaethical Perspectives on 'Benchmarking' AI Ethics","venue":"cs.CY","work_id":"fe834e84-2090-49e8-a636-fb7718dd4ae2","year":2022},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":62,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.020829Z"},"links":{"cited_paper":"/paper/2204.05151","citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:9da30b8271fb72565815f6bf4a91d36e3f02e64c2a02bfbb619a0fb0b166ecdb","observation_id":"93d63b08-f881-4019-832e-4f1cb0049438","resolution":{"observed_at":"2026-08-08T15:06:57.560951Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.12220","last_updated":"2024-10-30T12:14:35Z","snapshot_observed_at":"2026-08-16T13:33:35.862588Z","submitted_at":"2024-07-17T00:06:30Z","title":"Questionable practices in machine learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.12220","snapshot_observed_at":"2026-08-08T15:06:55.026059Z","title":"Vazquez, Niclas Kupper, Misha Yagudin, and Laurence Aitchison","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.026059Z"},"links":{"cited_paper":"/paper/2407.12220","citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:d075eb1da7e61ec09259566eb21c391beadef51655fa4ef3640af9d18c9a5b3d","observation_id":"bc32110d-779b-4610-9e2a-3f26bda585e5","resolution":{"observed_at":"2026-08-08T15:06:55.026059Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:55.031305Z","title":"Question and answer test-train overlap in open-domain question answering datasets","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.031305Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:0c3ceabca061312eae6609b70caaf48668d74cf71826c6698d5dfd5512aaa8bc","observation_id":"a20a4b3b-91a0-4c83-aa53-351a8cccc922","resolution":{"observed_at":"2026-08-08T15:06:55.031305Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2211.09110","last_updated":"2023-10-01T21:44:23Z","snapshot_observed_at":"2026-08-10T23:10:13.900680Z","submitted_at":"2022-11-16T18:51:34Z","title":"Holistic Evaluation of Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2211.09110","snapshot_observed_at":"2026-08-08T15:06:55.036681Z","title":"Manning, Christopher Ré, Diana Acosta-Navas, Drew A","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":65,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.036681Z"},"links":{"cited_paper":"/paper/2211.09110","citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:cef13d3c0bdd2f09893fdf6d1bcf3c0fc5be7f0356d93975f76c5ef91ae8e582","observation_id":"f7e73f1a-be3b-4574-869d-ebff3a8d7aaa","resolution":{"observed_at":"2026-08-08T15:06:55.036681Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03100","last_updated":"2025-01-31T14:59:17Z","snapshot_observed_at":"2026-08-17T06:04:45.297512Z","submitted_at":"2023-06-01T00:01:43Z","title":"Rethinking Model Evaluation as Narrowing the Socio-Technical Gap","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.03100","snapshot_observed_at":"2026-08-08T15:06:55.041880Z","title":"Vera Liao and Ziang Xiao","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":66,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.041880Z"},"links":{"cited_paper":"/paper/2306.03100","citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:f88623969023f25d2e30107cc4acbf8cf046a4c84d25c4bf8bfd3c15d8b9e8d4","observation_id":"b8bf805d-b04c-4813-a1ca-07f82e07f1f8","resolution":{"observed_at":"2026-08-08T15:06:55.041880Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:55.046898Z","title":"Are we learning yet? a meta review of evaluation failures across machine learning","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":67,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.046898Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:27467e11685a2efdf3913f57c9e8718d57fadf8d5baccda6a6e1bcfe7df57ae9","observation_id":"cf0f3579-3aa1-47b0-8995-0582f3e832a5","resolution":{"observed_at":"2026-08-08T15:06:55.046898Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2104.06387","last_updated":"2021-07-01T18:50:23Z","snapshot_observed_at":"2026-08-20T05:52:16.750270Z","submitted_at":"2021-04-13T17:45:50Z","title":"ExplainaBoard: An Explainable Leaderboard for NLP","version":2},"cited_work":{"arxiv_id":"2104.06387","doi":null,"metadata_source":"pith","pith_arxiv_id":"2104.06387","snapshot_observed_at":"2026-08-08T15:06:57.482106Z","title":"ExplainaBoard: An Explainable Leaderboard for NLP","venue":"cs.CL","work_id":"ffd98238-0ef0-45f0-9177-d3d33d717662","year":2021},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":68,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.051780Z"},"links":{"cited_paper":"/paper/2104.06387","citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:3e0430a5a89d5d46c175ca24712ea8f7c37cd35b22269c05f94ab09676bd2825","observation_id":"c509233e-65dc-496a-8ebd-1f0d0023ed99","resolution":{"observed_at":"2026-08-08T15:06:57.487779Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2024.23343","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:57.455906Z","title":"AI competitions as infrastructures of power in medical imaging","venue":null,"work_id":"4ce6c788-6d45-4d46-b49e-857e9909536d","year":2024},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":69,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.056702Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:fb5d8bbf1bcb91fc8d0b19c5a66077c3192e9ff257d3f07a6b81da7cbde9479b","observation_id":"d9afcc5e-d8e6-4f7d-bc63-44ebe7af9679","resolution":{"observed_at":"2026-08-08T15:06:57.464372Z","resolver_source":"raw_fallback","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.12649","last_updated":"2025-06-05T14:35:00Z","snapshot_observed_at":"2026-08-21T23:01:27.299707Z","submitted_at":"2024-02-20T01:49:15Z","title":"Bias in Language Models: Beyond Trick Tests and Toward RUTEd Evaluation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.12649","snapshot_observed_at":"2026-08-08T15:06:55.061220Z","title":"Bias in Language Models : Beyond Trick Tests and Toward RUTEd Evaluation , February 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":70,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.061220Z"},"links":{"cited_paper":"/paper/2402.12649","citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:9ebcf858891f40c6a28eddaa6045963f35373acd862de60950228290bc96120c","observation_id":"d9566273-dcf3-4b57-9649-4fa3175e583f","resolution":{"observed_at":"2026-08-08T15:06:55.061220Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:55.066279Z","title":"Data contamination: From memorization to exploitation","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":71,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.066279Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:c52471c2cb1957a2e83c4f3ea6b0a7ed3586e669d0c02202baed8d3f87ec15ca","observation_id":"63271f95-0e13-45ce-b248-98bbf8216ed6","resolution":{"observed_at":"2026-08-08T15:06:55.066279Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"0763.2023","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:57.362852Z","title":"Practices of Benchmarking : Vulnerability in the Computer Vision Pipeline","venue":null,"work_id":"f3fbb4c2-537a-4db7-8dd8-75bd242e3529","year":2023},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":72,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.071125Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:6287d135fbe816172b51ba8b30efacb3d6c89724fe84cb08a9d00fea957ea176","observation_id":"2c701896-0398-47f9-8beb-827cb86465bb","resolution":{"observed_at":"2026-08-08T15:06:57.371079Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"4446.12746","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:57.238658Z","title":"Put to the test: For a new sociology of testing","venue":null,"work_id":"ff981604-7ff0-4bc6-9d08-59cff4b791c1","year":2020},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":73,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.076699Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:767f9d060b893fe779d22c3331fb1495c268a87cefe0866b234deb8db53bd9b1","observation_id":"b10177e0-a44d-42b7-8306-ec8676ad9faf","resolution":{"observed_at":"2026-08-08T15:06:57.246799Z","resolver_source":"raw_fallback","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:55.081262Z","title":"Mlperf training benchmark","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":74,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.081262Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:1bdb65df26d42cf9d97fa807d960a7e807c8d12423b25edf000010346fcfa65c","observation_id":"46985aa9-a16f-4070-a1c6-89fdbd217fb9","resolution":{"observed_at":"2026-08-08T15:06:55.081262Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2020.29748","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:57.150414Z","title":"Mlperf: An industry standard benchmark suite for machine learning performance","venue":null,"work_id":"271ca16d-0fad-4694-ae04-359e07b80d5a","year":2020},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":75,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.085797Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:2b5ad3cbc64717f056f97ea98ccffa75071c06c02085b4a800fe1b54c9175fd3","observation_id":"c9f87e9b-4d73-437f-9f10-3e85c932022c","resolution":{"observed_at":"2026-08-08T15:06:57.158956Z","resolver_source":"raw_fallback","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.09880","last_updated":"2024-10-14T02:11:29Z","snapshot_observed_at":"2026-08-20T13:45:09.934614Z","submitted_at":"2024-02-15T11:08:10Z","title":"Inadequacies of Large Language Model Benchmarks in the Era of Generative Artificial Intelligence","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.09880","snapshot_observed_at":"2026-08-08T15:06:55.090364Z","title":"McIntosh, Teo Susnjak, Nalin Arachchilage, Tong Liu, Paul Watters, and Malka N","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":76,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.090364Z"},"links":{"cited_paper":"/paper/2402.09880","citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:5f562fa78cca339b1d6845ad0260cebffd0a5571f8762187d4cabb649b76c060","observation_id":"58a662b0-38db-4b24-8cf1-35d1af3a5c68","resolution":{"observed_at":"2026-08-08T15:06:55.090364Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.04984","last_updated":"2025-01-14T20:16:01Z","snapshot_observed_at":"2026-08-15T04:02:05.429942Z","submitted_at":"2024-12-06T12:09:50Z","title":"Frontier Models are Capable of In-context Scheming","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.04984","snapshot_observed_at":"2026-08-08T15:06:55.095295Z","title":"Frontier Models are Capable of In -context Scheming , December 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":77,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.095295Z"},"links":{"cited_paper":"/paper/2412.04984","citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:a908851c0cf2896806717d87e5620dd92d24fdf24380ed9f4a91928e59aae9fb","observation_id":"2a4c125f-88e8-42c1-9346-47ca4767b047","resolution":{"observed_at":"2026-08-08T15:06:55.095295Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2208.12852","last_updated":"2022-08-26T19:45:51Z","snapshot_observed_at":"2026-08-19T09:41:15.692637Z","submitted_at":"2022-08-26T19:45:51Z","title":"What Do NLP Researchers Believe? Results of the NLP Community Metasurvey","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2208.12852","snapshot_observed_at":"2026-08-08T15:06:55.100210Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":78,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.100210Z"},"links":{"cited_paper":"/paper/2208.12852","citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:96b353d5e94696bfd94e8347ee93137d6fbbc45efe19005e2514775cf8b45b12","observation_id":"23568cc2-de9c-48f6-aee7-377fb84f4349","resolution":{"observed_at":"2026-08-08T15:06:55.100210Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"9856.35828","doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:57.018430Z","title":"Benchmarking the Benchmarks","venue":null,"work_id":"f99afd92-aa30-4f1f-9cb6-a60d40c12559","year":2023},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":79,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.104848Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:0747345016e138daff40b051f744ebaff8e4edb7c4b76e18bcd63760a073a440","observation_id":"f504545f-ffa8-40bd-880e-6f80fcbc1095","resolution":{"observed_at":"2026-08-08T15:06:57.028571Z","resolver_source":"raw_fallback","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1126/science.adj595","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:55.541018Z","title":"How do we know how smart AI systems are? Science, 381 0 (6654), July 2023","venue":null,"work_id":"3e9a991e-f61a-4d71-974f-b4124bce80da","year":2023},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":80,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.109305Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:7c4fb7851e11c9ecf632817f79e3157b171314b30c3519d9329eb82e771a8968","observation_id":"ae171c1f-3071-4919-9e8e-09b4bf06203f","resolution":{"observed_at":"2026-08-08T15:06:55.546298Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.00595","last_updated":"2024-05-06T10:20:26Z","snapshot_observed_at":"2026-08-16T14:30:34.654390Z","submitted_at":"2023-12-31T22:21:36Z","title":"State of What Art? A Call for Multi-Prompt LLM Evaluation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.00595","snapshot_observed_at":"2026-08-08T15:06:55.118860Z","title":"State of What Art ? A Call for Multi - Prompt LLM Evaluation , May 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":82,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.118860Z"},"links":{"cited_paper":"/paper/2401.00595","citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:2d5014c7da6ebaaff5c0e5a539e9d8d140ab9e514e4bb881903476effdd9fc28","observation_id":"fff97b18-4589-4e30-a1df-0d7823c72029","resolution":{"observed_at":"2026-08-08T15:06:55.118860Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:55.123748Z","title":"Proxies: The Cultural Work of Standing In","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":83,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.123748Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:54cf19b6c06700a7edfce1901c005ed3144e8642ce7648a768359e71634cc27a","observation_id":"4428494f-5418-4eff-80bd-e7bd448a556a","resolution":{"observed_at":"2026-08-08T15:06:55.123748Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:55.128502Z","title":"GPT -4 and professional benchmarks: the wrong answer to the wrong question, March 2023 a","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":84,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.128502Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:7c38245458b2045a83f3fca22da0b663aeb263c5cb96310a0f635a453394f3f5","observation_id":"468fafed-c739-400b-972d-171bf0681ca3","resolution":{"observed_at":"2026-08-08T15:06:55.128502Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:55.133342Z","title":"Evaluating LLMs is a minefield, 2023 b","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":85,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.133342Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:c81a4da40de63d161da4980d45e07db1d3b2633035e71293dea0e8e742d4003b","observation_id":"a64299a0-a34b-4db1-86d4-fb60e0642331","resolution":{"observed_at":"2026-08-08T15:06:55.133342Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:55.138308Z","title":"Feder Cooper, Daphne Ippolito, Christopher A","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":86,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.138308Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:47b1e8a7a6098a5a57049ad73ffd22ceda0293335dd56afa70a1c79ac1e520bd","observation_id":"de525309-a9a4-45cd-9c6f-56b9e6af60b0","resolution":{"observed_at":"2026-08-08T15:06:55.138308Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17035","last_updated":"2023-11-28T18:47:03Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-11-28T18:47:03Z","title":"Scalable Extraction of Training Data from (Production) Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.17035","snapshot_observed_at":"2026-08-08T15:06:55.143422Z","title":"Feder Cooper, Daphne Ippolito, Christopher A","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":87,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.143422Z"},"links":{"cited_paper":"/paper/2311.17035","citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:4e064f6f211676cbd1b0a6b85c1a7330176dc84dfff5cfab477d7433f095aef2","observation_id":"cc0240f9-4c51-43b3-be97-98ed3201c4e5","resolution":{"observed_at":"2026-08-08T15:06:55.143422Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:55.149163Z","title":"Evaluating Hosting Provider Security Through Abuse Data and the Creation of Metrics","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":88,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.149163Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:07a3761ab8fd616ea641d24e35ea048f13215d22e0c2172005ad8b0de18f82ce","observation_id":"58fab166-1b30-442f-875e-7a55e2d94f62","resolution":{"observed_at":"2026-08-08T15:06:55.149163Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1909.12475","last_updated":"2019-11-15T09:44:33Z","snapshot_observed_at":"2026-08-13T15:00:07.295132Z","submitted_at":"2019-09-27T02:42:58Z","title":"Hidden Stratification Causes Clinically Meaningful Failures in Machine Learning for Medical Imaging","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1909.12475","snapshot_observed_at":"2026-08-08T15:06:55.154615Z","title":"Hidden Stratification Causes Clinically Meaningful Failures in Machine Learning for Medical Imaging , November 2019","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":89,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.154615Z"},"links":{"cited_paper":"/paper/1909.12475","citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:0fc648e6d9e013bec3da1912013107b6a3797cac9d2cb4bafda656da22a110e1","observation_id":"ac601b34-b7c9-4d6d-835d-4340cd3a77fa","resolution":{"observed_at":"2026-08-08T15:06:55.154615Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.17861","last_updated":"2025-02-27T20:24:21Z","snapshot_observed_at":"2026-08-16T19:12:08.563193Z","submitted_at":"2024-02-27T19:52:54Z","title":"Towards AI Accountability Infrastructure: Gaps and Opportunities in AI Audit Tooling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.17861","snapshot_observed_at":"2026-08-08T15:06:55.159834Z","title":"Towards AI Accountability Infrastructure : Gaps and Opportunities in AI Audit Tooling , March 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":90,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.159834Z"},"links":{"cited_paper":"/paper/2402.17861","citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:77b7ab6bd3365d2a1869e674c97ff31f7d4584fd68c62ebc3120a10fc9b31208","observation_id":"680ee922-a081-4184-bcc1-ab836511d940","resolution":{"observed_at":"2026-08-08T15:06:55.159834Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:55.164878Z","title":"The social construction of datasets: On the practices, processes, and challenges of dataset creation for machine learning","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":91,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.164878Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:7b53e93e94eb7b72e2210de72493fcb60c232ad4feca4e5b056f35db64e80f11","observation_id":"f56a2430-b014-4f9f-bff4-86196156ea69","resolution":{"observed_at":"2026-08-08T15:06:55.164878Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.00252","last_updated":"2024-08-30T20:52:19Z","snapshot_observed_at":"2026-08-16T13:22:31.437866Z","submitted_at":"2024-08-30T20:52:19Z","title":"Building Better Datasets: Seven Recommendations for Responsible Design from Dataset Creators","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.00252","snapshot_observed_at":"2026-08-08T15:06:55.169494Z","title":"Building Better Datasets : Seven Recommendations for Responsible Design from Dataset Creators , August 2024 b","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":92,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.169494Z"},"links":{"cited_paper":"/paper/2409.00252","citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:f4b6e131c327a6577bd0b84da4d4e2c313f6086277f49d71bac22352fd807a80","observation_id":"00f285e6-ae51-43a5-a5ee-a27270e80710","resolution":{"observed_at":"2026-08-08T15:06:55.169494Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:55.174479Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":93,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.174479Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:b15cae2a58b56a0b2dcb8c915d86b6e3d01fc1c8bf2527333d3f184d9c09e237","observation_id":"5dbe9932-8ea6-4758-9406-82d93360e41a","resolution":{"observed_at":"2026-08-08T15:06:55.174479Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:55.178850Z","title":"Mapping global dynamics of benchmark creation and saturation in artificial intelligence","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":94,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.178850Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:fedee00e658be3a24f8a7243128324d89c502f16263bbe17df2c8c226f6756da","observation_id":"72053b20-d81c-4e13-869b-b3e84aa7b380","resolution":{"observed_at":"2026-08-08T15:06:55.178850Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:55.183830Z","title":"Benchmark","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":95,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.183830Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:2f5ecc7210754fa0191b838f3ae0662c68c100a26691fbee7009fff233e14a5c","observation_id":"e0564aa4-5cb6-4694-8454-4fc8f9af7fc4","resolution":{"observed_at":"2026-08-08T15:06:55.183830Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.11672","last_updated":"2024-10-15T15:05:41Z","snapshot_observed_at":"2026-08-16T13:09:09.333038Z","submitted_at":"2024-10-15T15:05:41Z","title":"Leaving the barn door open for Clever Hans: Simple features predict LLM benchmark answers","version":1},"cited_work":{"arxiv_id":"2410.11672","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.11672","snapshot_observed_at":"2026-08-08T15:06:56.769040Z","title":"Leaving the barn door open for Clever Hans: Simple features predict LLM benchmark answers","venue":"cs.CL","work_id":"5cfa1b92-ffa3-469f-827d-e0681094cf94","year":2024},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":96,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.188523Z"},"links":{"cited_paper":"/paper/2410.11672","citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:287d1097c309401359bacd62afbc698169fe486af67c9801fca1b8294bf8a63d","observation_id":"d7b56158-c7e9-4246-b35f-dd60e74dd4c9","resolution":{"observed_at":"2026-08-08T15:06:56.775567Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2022.nlppower-1.1","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"Raison d’être of the benchmark dataset: A Survey of Current Practices of Benchmark Dataset Sharing Platforms","venue":null,"work_id":"8951bd2a-d16c-4864-a531-9e07cbb181ad","year":2022},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":97,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.193886Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:e2372ca3c8e9828510e01a7f0adfa56eed07938bd715611581e68da1562c8090","observation_id":"e183a41a-e920-47e9-8a48-c2532bcbb09c","resolution":{"observed_at":"2026-08-08T15:06:55.518509Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:55.198904Z","title":"Bender, Emily Denton, and Alex Hanna","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":98,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.198904Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:4f8924102b505f6eece72982696e673f04e777d200260e610ccd96aea2dc3791","observation_id":"b321f8a4-831b-4f54-a9e0-8ea230480463","resolution":{"observed_at":"2026-08-08T15:06:55.198904Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.07458","last_updated":"2025-01-13T16:28:01Z","snapshot_observed_at":"2026-08-10T20:39:12.747933Z","submitted_at":"2025-01-13T16:28:01Z","title":"Understanding and Benchmarking Artificial Intelligence: OpenAI's o3 Is Not AGI","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.07458","snapshot_observed_at":"2026-08-08T15:06:55.203597Z","title":"Understanding and Benchmarking Artificial Intelligence : OpenAI 's o3 Is Not AGI , January 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":99,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.203597Z"},"links":{"cited_paper":"/paper/2501.07458","citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:71810252d3b96fc46c2727fac0495b61139e5eb7ac723b851212aa82e99ca2f4","observation_id":"a472054a-1414-4ad1-b0a1-652e7056d5b0","resolution":{"observed_at":"2026-08-08T15:06:55.203597Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1177/016224399301800103","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-10T05:30:23.456663Z","title":"Testing - One , Two , Three ... Testing !","venue":"Science Technology & Human Values","work_id":"710eb2ff-dce6-45aa-bfcb-8fda89173839","year":1993},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":100,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.208380Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:7ee5296e061b1d63d0ed4098b634591f82670ee596cf2e7cd02ce5b5c1ae07b3","observation_id":"b65cb9b1-d684-4395-a2ad-fc320d5d3399","resolution":{"observed_at":"2026-08-08T15:06:55.501858Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.08392","last_updated":"2024-12-11T14:02:55Z","snapshot_observed_at":"2026-08-18T14:21:55.078510Z","submitted_at":"2024-12-11T14:02:55Z","title":"The Roles of English in Evaluating Multilingual Language Models","version":1},"cited_work":{"arxiv_id":"2412.08392","doi":null,"metadata_source":"pith","pith_arxiv_id":"2412.08392","snapshot_observed_at":"2026-08-08T15:06:56.620307Z","title":"The Roles of English in Evaluating Multilingual Language Models","venue":"cs.CL","work_id":"53f6ce24-1d36-460e-8e44-3171b6d0e58c","year":2024},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":101,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.213500Z"},"links":{"cited_paper":"/paper/2412.08392","citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:d6771a0e617f5d4587a6cb0986712a09e66f5c45e65284e7b040f5c99a1dcdf0","observation_id":"22c55187-dd1c-467d-8772-e84504bdcd3a","resolution":{"observed_at":"2026-08-08T15:06:56.626299Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03693","last_updated":"2023-10-05T17:12:17Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-05T17:12:17Z","title":"Fine-tuning Aligned Language Models Compromises Safety, Even When Users Do Not Intend To!","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03693","snapshot_observed_at":"2026-08-08T15:06:55.218219Z","title":"Fine-tuning Aligned Language Models Compromises Safety , Even When Users Do Not Intend To !, October 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":102,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.218219Z"},"links":{"cited_paper":"/paper/2310.03693","citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:7085be73b796f1b90b0cfeeafbff0a8d804f7ec3c6886eb86e6fc84e73336580","observation_id":"b9c3d3e6-c9e5-4bb0-be05-456267fe790b","resolution":{"observed_at":"2026-08-08T15:06:55.218219Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:55.223108Z","title":"Bender, Alex Hanna, and Amandalynne Paullada","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":103,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.223108Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:bf9c7bd9fde80d92c453c0097644f720f485f35fdc4f5805b36d0bd80204236e","observation_id":"c4dc11da-f18e-4b8c-9f60-12df1bb61f7b","resolution":{"observed_at":"2026-08-08T15:06:55.223108Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T15:06:55.228037Z","title":"Gaps in the Safety Evaluation of Generative AI","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"reference_index":104,"source":"arxiv_source","source_observed_at":"2026-08-08T15:06:55.228037Z"},"links":{"citing_paper":"/paper/2502.06559"},"observation_digest":"sha256:b581012dca95e8091517e31b90e46d98c8d9dbc2da64fc2c6e368dd333e0812d","observation_id":"534bbdae-29cd-496d-8277-abbdc915d5f0","resolution":{"observed_at":"2026-08-08T15:06:55.228037Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","latest_version":2,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":0,"metadata_mismatch":5,"parse_uncertain":0,"unresolved":77,"verified_exact":18,"verified_fuzzy":0},"total_outbound_references":135},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"thesis":"As of 23 August 2026, this Paper Citation Record lists 100 of 135 outbound references and 33 inbound Pith citation observations for arXiv:2502.06559."}