{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:C3N7F6T4K7IHMSE7V57P5RAEZU","short_pith_number":"pith:C3N7F6T4","schema_version":"1.0","canonical_sha256":"16dbf2fa7c57d076489faf7efec404cd131eca990d25c2be784ec53fd7deb2c2","source":{"kind":"arxiv","id":"2412.08653","version":1},"attestation_state":"computed","paper":{"title":"What AI evaluations for preventing catastrophic risks can and cannot do","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CY","authors_text":"Lisa Thiergart, Peter Barnett","submitted_at":"2024-11-26T18:00:36Z","abstract_excerpt":"AI evaluations are an important component of the AI governance toolkit, underlying current approaches to safety cases for preventing catastrophic risks. Our paper examines what these evaluations can and cannot tell us. Evaluations can establish lower bounds on AI capabilities and assess certain misuse risks given sufficient effort from evaluators.\n  Unfortunately, evaluations face fundamental limitations that cannot be overcome within the current paradigm. These include an inability to establish upper bounds on capabilities, reliably forecast future model capabilities, or robustly assess risks"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.08653","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CY","submitted_at":"2024-11-26T18:00:36Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"b8e7ca299f97a08046c6c38b96dd445856276415ac46054cc59c807c54dacd67","abstract_canon_sha256":"beae1739d19c3ef470774bfe59ed1d197ff339afb16b505340490ea63e9ff338"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:47:58.609440Z","signature_b64":"gLmaxiA5NiCVgym6U0x7qbOP1z7owjGo8w0wOI9bIBPGT+6TmFCLLUPL6RA/Qum0cgMnfCzGrLf32Lh38/EeDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"16dbf2fa7c57d076489faf7efec404cd131eca990d25c2be784ec53fd7deb2c2","last_reissued_at":"2026-07-05T09:47:58.608963Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:47:58.608963Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"What AI evaluations for preventing catastrophic risks can and cannot do","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CY","authors_text":"Lisa Thiergart, Peter Barnett","submitted_at":"2024-11-26T18:00:36Z","abstract_excerpt":"AI evaluations are an important component of the AI governance toolkit, underlying current approaches to safety cases for preventing catastrophic risks. Our paper examines what these evaluations can and cannot tell us. Evaluations can establish lower bounds on AI capabilities and assess certain misuse risks given sufficient effort from evaluators.\n  Unfortunately, evaluations face fundamental limitations that cannot be overcome within the current paradigm. These include an inability to establish upper bounds on capabilities, reliably forecast future model capabilities, or robustly assess risks"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.08653","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.08653/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.08653","created_at":"2026-07-05T09:47:58.609021+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.08653v1","created_at":"2026-07-05T09:47:58.609021+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.08653","created_at":"2026-07-05T09:47:58.609021+00:00"},{"alias_kind":"pith_short_12","alias_value":"C3N7F6T4K7IH","created_at":"2026-07-05T09:47:58.609021+00:00"},{"alias_kind":"pith_short_16","alias_value":"C3N7F6T4K7IHMSE7","created_at":"2026-07-05T09:47:58.609021+00:00"},{"alias_kind":"pith_short_8","alias_value":"C3N7F6T4","created_at":"2026-07-05T09:47:58.609021+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.08529","citing_title":"Scaffold Effects on GAIA: A Controlled Comparison","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28694","citing_title":"Verifying Restrictions on Frontier AI Research","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14070","citing_title":"From Disclosure to Self-Referential Opacity: Six Dimensions of Strain in Current AI Governance","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/C3N7F6T4K7IHMSE7V57P5RAEZU","json":"https://pith.science/pith/C3N7F6T4K7IHMSE7V57P5RAEZU.json","graph_json":"https://pith.science/api/pith-number/C3N7F6T4K7IHMSE7V57P5RAEZU/graph.json","events_json":"https://pith.science/api/pith-number/C3N7F6T4K7IHMSE7V57P5RAEZU/events.json","paper":"https://pith.science/paper/C3N7F6T4"},"agent_actions":{"view_html":"https://pith.science/pith/C3N7F6T4K7IHMSE7V57P5RAEZU","download_json":"https://pith.science/pith/C3N7F6T4K7IHMSE7V57P5RAEZU.json","view_paper":"https://pith.science/paper/C3N7F6T4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.08653&json=true","fetch_graph":"https://pith.science/api/pith-number/C3N7F6T4K7IHMSE7V57P5RAEZU/graph.json","fetch_events":"https://pith.science/api/pith-number/C3N7F6T4K7IHMSE7V57P5RAEZU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/C3N7F6T4K7IHMSE7V57P5RAEZU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/C3N7F6T4K7IHMSE7V57P5RAEZU/action/storage_attestation","attest_author":"https://pith.science/pith/C3N7F6T4K7IHMSE7V57P5RAEZU/action/author_attestation","sign_citation":"https://pith.science/pith/C3N7F6T4K7IHMSE7V57P5RAEZU/action/citation_signature","submit_replication":"https://pith.science/pith/C3N7F6T4K7IHMSE7V57P5RAEZU/action/replication_record"}},"created_at":"2026-07-05T09:47:58.609021+00:00","updated_at":"2026-07-05T09:47:58.609021+00:00"}