{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:F3YS53CACUZ2CJRCGZO7L4AEUC","short_pith_number":"pith:F3YS53CA","schema_version":"1.0","canonical_sha256":"2ef12eec401533a12622365df5f004a0b9fed8872c601196bbe394dbb575b7ae","source":{"kind":"arxiv","id":"2305.15324","version":2},"attestation_state":"computed","paper":{"title":"Model evaluation for extreme risks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Allan Dafoe, Been Kim, Ben Garfinkel, Daniel Kokotajlo, Divya Siddarth, Iason Gabriel, Jack Clark, Jade Leung, Jess Whittlestone, Lewis Ho, Markus Anderljung, Mary Phuong, Nahema Marchal, Noam Kolt, Paul Christiano, Sebastian Farquhar, Shahar Avin, Toby Shevlane, Vijay Bolina, Will Hawkins, Yoshua Bengio","submitted_at":"2023-05-24T16:38:43Z","abstract_excerpt":"Current approaches to building general-purpose AI systems tend to produce systems with both beneficial and harmful capabilities. Further progress in AI development could lead to capabilities that pose extreme risks, such as offensive cyber capabilities or strong manipulation skills. We explain why model evaluation is critical for addressing extreme risks. Developers must be able to identify dangerous capabilities (through \"dangerous capability evaluations\") and the propensity of models to apply their capabilities for harm (through \"alignment evaluations\"). These evaluations will become critica"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.15324","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2023-05-24T16:38:43Z","cross_cats_sorted":[],"title_canon_sha256":"3d332188b9141337555a26584723aa00326b01e10ce4964948c57eaf5d2cba5b","abstract_canon_sha256":"6f314c862888c36771598fa2a0f5405fab6b81bca5de34835a53b0cd7f5d7bd0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:53:38.172631Z","signature_b64":"Vk5E8KmUqW0LwruOP3qTVtDbAKPcPvKX9cTueLCOAZxO8U4wfEGuTWdA0mSaFkjvCS76I2PtZHAmoWcJROVmCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2ef12eec401533a12622365df5f004a0b9fed8872c601196bbe394dbb575b7ae","last_reissued_at":"2026-07-05T06:53:38.172155Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:53:38.172155Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Model evaluation for extreme risks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Allan Dafoe, Been Kim, Ben Garfinkel, Daniel Kokotajlo, Divya Siddarth, Iason Gabriel, Jack Clark, Jade Leung, Jess Whittlestone, Lewis Ho, Markus Anderljung, Mary Phuong, Nahema Marchal, Noam Kolt, Paul Christiano, Sebastian Farquhar, Shahar Avin, Toby Shevlane, Vijay Bolina, Will Hawkins, Yoshua Bengio","submitted_at":"2023-05-24T16:38:43Z","abstract_excerpt":"Current approaches to building general-purpose AI systems tend to produce systems with both beneficial and harmful capabilities. Further progress in AI development could lead to capabilities that pose extreme risks, such as offensive cyber capabilities or strong manipulation skills. We explain why model evaluation is critical for addressing extreme risks. Developers must be able to identify dangerous capabilities (through \"dangerous capability evaluations\") and the propensity of models to apply their capabilities for harm (through \"alignment evaluations\"). These evaluations will become critica"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.15324","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.15324/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.15324","created_at":"2026-07-05T06:53:38.172214+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.15324v2","created_at":"2026-07-05T06:53:38.172214+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.15324","created_at":"2026-07-05T06:53:38.172214+00:00"},{"alias_kind":"pith_short_12","alias_value":"F3YS53CACUZ2","created_at":"2026-07-05T06:53:38.172214+00:00"},{"alias_kind":"pith_short_16","alias_value":"F3YS53CACUZ2CJRC","created_at":"2026-07-05T06:53:38.172214+00:00"},{"alias_kind":"pith_short_8","alias_value":"F3YS53CA","created_at":"2026-07-05T06:53:38.172214+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":30,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18532","citing_title":"AI Sandboxes: A Threat Model, Taxonomy, and Measurement Framework","ref_index":142,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01854","citing_title":"Has This Checkpoint Been Abliterated? A Two-Signal Audit and Its Failure Map","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09809","citing_title":"Evaluation Cards: An Interpretive Layer for AI Evaluation Reporting","ref_index":97,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05330","citing_title":"A Model of Multi-turn Human Persuadability Using Probabilistic Belief Tracing","ref_index":114,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15164","citing_title":"Position: Behavioural Assurance Cannot Verify the Safety Claims Governance Now Demands","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02211","citing_title":"Consistency Training while Mitigating Obfuscation via Rate Matching","ref_index":138,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06286","citing_title":"LLMs Can Leak Training Data But Do They Want To? A Propensity-Aware Evaluation of Memorization in LLMs","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2305.09620","citing_title":"AI-Augmented Surveys: Leveraging Large Language Models and Surveys for Opinion Prediction","ref_index":91,"is_internal_anchor":false},{"citing_arxiv_id":"2312.11805","citing_title":"Gemini: A Family of Highly Capable Multimodal Models","ref_index":95,"is_internal_anchor":false},{"citing_arxiv_id":"2512.05742","citing_title":"Internal Deployment in the AI Act","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21095","citing_title":"Backchaining Loss of Control Mitigations from Mission-Specific Benchmarks in National Security","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21095","citing_title":"Backchaining Loss of Control Mitigations from Mission-Specific Benchmarks in National Security","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19722","citing_title":"Measuring Safety Alignment Effects in Autonomous Security Agents","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2506.07586","citing_title":"MalGEN: A Testbed for Modeling and Evaluating Malware Behaviors","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2506.06414","citing_title":"Benchmarking Misuse Mitigation Against Covert Adversaries","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2401.05561","citing_title":"TrustLLM: Trustworthiness in Large Language Models","ref_index":178,"is_internal_anchor":false},{"citing_arxiv_id":"2511.05914","citing_title":"Designing Incident Reporting Systems for Harms from General-Purpose AI","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12288","citing_title":"TokenRatio: Principled Token-Level Preference Optimization via Ratio Matching","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12199","citing_title":"Overtrained, Not Misaligned","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12288","citing_title":"TokenRatio: Principled Token-Level Preference Optimization via Ratio Matching","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25119","citing_title":"Evaluation without Generation: Non-Generative Assessment of Harmful Model Specialization with Applications to CSAM","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24966","citing_title":"Risk Reporting for Developers' Internal AI Model Use","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01420","citing_title":"Artificial Jagged Intelligence as Uneven Optimization Energy Allocation Capability Concentration, Redistribution, and Optimization Governance","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21769","citing_title":"Who Defines \"Best\"? Towards Interactive, User-Defined Evaluation of LLM Leaderboards","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2502.18864","citing_title":"Towards an AI co-scientist","ref_index":35,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/F3YS53CACUZ2CJRCGZO7L4AEUC","json":"https://pith.science/pith/F3YS53CACUZ2CJRCGZO7L4AEUC.json","graph_json":"https://pith.science/api/pith-number/F3YS53CACUZ2CJRCGZO7L4AEUC/graph.json","events_json":"https://pith.science/api/pith-number/F3YS53CACUZ2CJRCGZO7L4AEUC/events.json","paper":"https://pith.science/paper/F3YS53CA"},"agent_actions":{"view_html":"https://pith.science/pith/F3YS53CACUZ2CJRCGZO7L4AEUC","download_json":"https://pith.science/pith/F3YS53CACUZ2CJRCGZO7L4AEUC.json","view_paper":"https://pith.science/paper/F3YS53CA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.15324&json=true","fetch_graph":"https://pith.science/api/pith-number/F3YS53CACUZ2CJRCGZO7L4AEUC/graph.json","fetch_events":"https://pith.science/api/pith-number/F3YS53CACUZ2CJRCGZO7L4AEUC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/F3YS53CACUZ2CJRCGZO7L4AEUC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/F3YS53CACUZ2CJRCGZO7L4AEUC/action/storage_attestation","attest_author":"https://pith.science/pith/F3YS53CACUZ2CJRCGZO7L4AEUC/action/author_attestation","sign_citation":"https://pith.science/pith/F3YS53CACUZ2CJRCGZO7L4AEUC/action/citation_signature","submit_replication":"https://pith.science/pith/F3YS53CACUZ2CJRCGZO7L4AEUC/action/replication_record"}},"created_at":"2026-07-05T06:53:38.172214+00:00","updated_at":"2026-07-05T06:53:38.172214+00:00"}