{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:IJQ5HV5P6JYSYWSQUVMLDESIB2","short_pith_number":"pith:IJQ5HV5P","schema_version":"1.0","canonical_sha256":"4261d3d7aff2712c5a50a558b192480e938334d366b26987a159e9f80f7f581b","source":{"kind":"arxiv","id":"2411.12820","version":1},"attestation_state":"computed","paper":{"title":"Declare and Justify: Explicit assumptions in AI evaluations are necessary for effective regulation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CY"],"primary_cat":"cs.AI","authors_text":"Lisa Thiergart, Peter Barnett","submitted_at":"2024-11-19T19:13:56Z","abstract_excerpt":"As AI systems advance, AI evaluations are becoming an important pillar of regulations for ensuring safety. We argue that such regulation should require developers to explicitly identify and justify key underlying assumptions about evaluations as part of their case for safety. We identify core assumptions in AI evaluations (both for evaluating existing models and forecasting future models), such as comprehensive threat modeling, proxy task validity, and adequate capability elicitation. Many of these assumptions cannot currently be well justified. If regulation is to be based on evaluations, it "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.12820","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-11-19T19:13:56Z","cross_cats_sorted":["cs.CY"],"title_canon_sha256":"c31cf31568663b43467d4cc454cd0af5a9ce988bcd5a96881248521e60ce8202","abstract_canon_sha256":"add960364ff10e1195c9d6b818e9c0d16c6978cebc1430b198112aa893d1d307"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:38:00.787138Z","signature_b64":"ZNQgzdLonXqhPRYLU7C4XPuSfxZjf9Ar/GpJHtJEPtT6rpwtRWksfH4qmuMmubZ5wtv06s6BC49Yaf5iRs1cBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4261d3d7aff2712c5a50a558b192480e938334d366b26987a159e9f80f7f581b","last_reissued_at":"2026-07-05T09:38:00.786658Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:38:00.786658Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Declare and Justify: Explicit assumptions in AI evaluations are necessary for effective regulation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CY"],"primary_cat":"cs.AI","authors_text":"Lisa Thiergart, Peter Barnett","submitted_at":"2024-11-19T19:13:56Z","abstract_excerpt":"As AI systems advance, AI evaluations are becoming an important pillar of regulations for ensuring safety. We argue that such regulation should require developers to explicitly identify and justify key underlying assumptions about evaluations as part of their case for safety. We identify core assumptions in AI evaluations (both for evaluating existing models and forecasting future models), such as comprehensive threat modeling, proxy task validity, and adequate capability elicitation. Many of these assumptions cannot currently be well justified. If regulation is to be based on evaluations, it "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.12820","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.12820/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.12820","created_at":"2026-07-05T09:38:00.786714+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.12820v1","created_at":"2026-07-05T09:38:00.786714+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.12820","created_at":"2026-07-05T09:38:00.786714+00:00"},{"alias_kind":"pith_short_12","alias_value":"IJQ5HV5P6JYS","created_at":"2026-07-05T09:38:00.786714+00:00"},{"alias_kind":"pith_short_16","alias_value":"IJQ5HV5P6JYSYWSQ","created_at":"2026-07-05T09:38:00.786714+00:00"},{"alias_kind":"pith_short_8","alias_value":"IJQ5HV5P","created_at":"2026-07-05T09:38:00.786714+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.18213","citing_title":"A Conceptual Framework for AI Capability Evaluations","ref_index":3,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IJQ5HV5P6JYSYWSQUVMLDESIB2","json":"https://pith.science/pith/IJQ5HV5P6JYSYWSQUVMLDESIB2.json","graph_json":"https://pith.science/api/pith-number/IJQ5HV5P6JYSYWSQUVMLDESIB2/graph.json","events_json":"https://pith.science/api/pith-number/IJQ5HV5P6JYSYWSQUVMLDESIB2/events.json","paper":"https://pith.science/paper/IJQ5HV5P"},"agent_actions":{"view_html":"https://pith.science/pith/IJQ5HV5P6JYSYWSQUVMLDESIB2","download_json":"https://pith.science/pith/IJQ5HV5P6JYSYWSQUVMLDESIB2.json","view_paper":"https://pith.science/paper/IJQ5HV5P","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.12820&json=true","fetch_graph":"https://pith.science/api/pith-number/IJQ5HV5P6JYSYWSQUVMLDESIB2/graph.json","fetch_events":"https://pith.science/api/pith-number/IJQ5HV5P6JYSYWSQUVMLDESIB2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IJQ5HV5P6JYSYWSQUVMLDESIB2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IJQ5HV5P6JYSYWSQUVMLDESIB2/action/storage_attestation","attest_author":"https://pith.science/pith/IJQ5HV5P6JYSYWSQUVMLDESIB2/action/author_attestation","sign_citation":"https://pith.science/pith/IJQ5HV5P6JYSYWSQUVMLDESIB2/action/citation_signature","submit_replication":"https://pith.science/pith/IJQ5HV5P6JYSYWSQUVMLDESIB2/action/replication_record"}},"created_at":"2026-07-05T09:38:00.786714+00:00","updated_at":"2026-07-05T09:38:00.786714+00:00"}