{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:FQO5XUSVPRWTC5NNWKNQ6M4LNI","short_pith_number":"pith:FQO5XUSV","schema_version":"1.0","canonical_sha256":"2c1ddbd2557c6d3175adb29b0f338b6a047da317b703aca08dccc7cb0b9a81f0","source":{"kind":"arxiv","id":"2407.09221","version":1},"attestation_state":"computed","paper":{"title":"Evaluating AI Evaluation: Perils and Prospects","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CY"],"primary_cat":"cs.AI","authors_text":"John Burden","submitted_at":"2024-07-12T12:37:13Z","abstract_excerpt":"As AI systems appear to exhibit ever-increasing capability and generality, assessing their true potential and safety becomes paramount. This paper contends that the prevalent evaluation methods for these systems are fundamentally inadequate, heightening the risks and potential hazards associated with AI. I argue that a reformation is required in the way we evaluate AI systems and that we should look towards cognitive sciences for inspiration in our approaches, which have a longstanding tradition of assessing general intelligence across diverse species. We will identify some of the difficulties"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.09221","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2024-07-12T12:37:13Z","cross_cats_sorted":["cs.CY"],"title_canon_sha256":"763ac05d1571a9344a020beec9b4d32e8f10f2e80cd0a986474d5ce72d2cc1d8","abstract_canon_sha256":"a17f5c304ce1f30ce877b399d3afa3a99851793e8dd6cfc3300121c53403260e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:43:16.460455Z","signature_b64":"ByH5Srj55i/+6ZzUGJrtZmLB0s96iJnizmGzbemY+Rdgg0Hr36FRfhr/wghZyK7ki8T4mooX6Ie70Ev3iRFhCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2c1ddbd2557c6d3175adb29b0f338b6a047da317b703aca08dccc7cb0b9a81f0","last_reissued_at":"2026-07-05T08:43:16.459930Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:43:16.459930Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluating AI Evaluation: Perils and Prospects","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CY"],"primary_cat":"cs.AI","authors_text":"John Burden","submitted_at":"2024-07-12T12:37:13Z","abstract_excerpt":"As AI systems appear to exhibit ever-increasing capability and generality, assessing their true potential and safety becomes paramount. This paper contends that the prevalent evaluation methods for these systems are fundamentally inadequate, heightening the risks and potential hazards associated with AI. I argue that a reformation is required in the way we evaluate AI systems and that we should look towards cognitive sciences for inspiration in our approaches, which have a longstanding tradition of assessing general intelligence across diverse species. We will identify some of the difficulties"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.09221","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.09221/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.09221","created_at":"2026-07-05T08:43:16.459994+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.09221v1","created_at":"2026-07-05T08:43:16.459994+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.09221","created_at":"2026-07-05T08:43:16.459994+00:00"},{"alias_kind":"pith_short_12","alias_value":"FQO5XUSVPRWT","created_at":"2026-07-05T08:43:16.459994+00:00"},{"alias_kind":"pith_short_16","alias_value":"FQO5XUSVPRWTC5NN","created_at":"2026-07-05T08:43:16.459994+00:00"},{"alias_kind":"pith_short_8","alias_value":"FQO5XUSV","created_at":"2026-07-05T08:43:16.459994+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.18213","citing_title":"A Conceptual Framework for AI Capability Evaluations","ref_index":9,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FQO5XUSVPRWTC5NNWKNQ6M4LNI","json":"https://pith.science/pith/FQO5XUSVPRWTC5NNWKNQ6M4LNI.json","graph_json":"https://pith.science/api/pith-number/FQO5XUSVPRWTC5NNWKNQ6M4LNI/graph.json","events_json":"https://pith.science/api/pith-number/FQO5XUSVPRWTC5NNWKNQ6M4LNI/events.json","paper":"https://pith.science/paper/FQO5XUSV"},"agent_actions":{"view_html":"https://pith.science/pith/FQO5XUSVPRWTC5NNWKNQ6M4LNI","download_json":"https://pith.science/pith/FQO5XUSVPRWTC5NNWKNQ6M4LNI.json","view_paper":"https://pith.science/paper/FQO5XUSV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.09221&json=true","fetch_graph":"https://pith.science/api/pith-number/FQO5XUSVPRWTC5NNWKNQ6M4LNI/graph.json","fetch_events":"https://pith.science/api/pith-number/FQO5XUSVPRWTC5NNWKNQ6M4LNI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FQO5XUSVPRWTC5NNWKNQ6M4LNI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FQO5XUSVPRWTC5NNWKNQ6M4LNI/action/storage_attestation","attest_author":"https://pith.science/pith/FQO5XUSVPRWTC5NNWKNQ6M4LNI/action/author_attestation","sign_citation":"https://pith.science/pith/FQO5XUSVPRWTC5NNWKNQ6M4LNI/action/citation_signature","submit_replication":"https://pith.science/pith/FQO5XUSVPRWTC5NNWKNQ6M4LNI/action/replication_record"}},"created_at":"2026-07-05T08:43:16.459994+00:00","updated_at":"2026-07-05T08:43:16.459994+00:00"}