{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:3EYWV6OGQQKRZM7UVQAZNVMOUG","short_pith_number":"pith:3EYWV6OG","schema_version":"1.0","canonical_sha256":"d9316af9c684151cb3f4ac0196d58ea1a24acd55dc6836777a510b6da1bf0ea9","source":{"kind":"arxiv","id":"2504.05259","version":1},"attestation_state":"computed","paper":{"title":"How to evaluate control measures for LLM agents? A trajectory from today to superintelligence","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CR"],"primary_cat":"cs.AI","authors_text":"Buck Shlegeris, Geoffrey Irving, Mikita Balesni, Tomek Korbak","submitted_at":"2025-04-07T16:52:52Z","abstract_excerpt":"As LLM agents grow more capable of causing harm autonomously, AI developers will rely on increasingly sophisticated control measures to prevent possibly misaligned agents from causing harm. AI developers could demonstrate that their control measures are sufficient by running control evaluations: testing exercises in which a red team produces agents that try to subvert control measures. To ensure control evaluations accurately capture misalignment risks, the affordances granted to this red team should be adapted to the capability profiles of the agents to be deployed under control measures.\n  I"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.05259","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-04-07T16:52:52Z","cross_cats_sorted":["cs.CR"],"title_canon_sha256":"700ee3ec79b29227f0172451c09c61c72bc024f85b9d4a159455cfcbe2b358b8","abstract_canon_sha256":"428df97f2968cdd3a4241645da0306cda312a14da3e75d80792d6464038f745b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:45:36.442929Z","signature_b64":"sy2VSiwlMeYc/p11f9W0QWG4Kq/YQzttoVtIzQr6W3lZzo9iev6RR+1cnpC/bIKVSKyTVLmlHt+lYsWwRTjnBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d9316af9c684151cb3f4ac0196d58ea1a24acd55dc6836777a510b6da1bf0ea9","last_reissued_at":"2026-07-05T10:45:36.442374Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:45:36.442374Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"How to evaluate control measures for LLM agents? A trajectory from today to superintelligence","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CR"],"primary_cat":"cs.AI","authors_text":"Buck Shlegeris, Geoffrey Irving, Mikita Balesni, Tomek Korbak","submitted_at":"2025-04-07T16:52:52Z","abstract_excerpt":"As LLM agents grow more capable of causing harm autonomously, AI developers will rely on increasingly sophisticated control measures to prevent possibly misaligned agents from causing harm. AI developers could demonstrate that their control measures are sufficient by running control evaluations: testing exercises in which a red team produces agents that try to subvert control measures. To ensure control evaluations accurately capture misalignment risks, the affordances granted to this red team should be adapted to the capability profiles of the agents to be deployed under control measures.\n  I"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.05259","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.05259/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.05259","created_at":"2026-07-05T10:45:36.442437+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.05259v1","created_at":"2026-07-05T10:45:36.442437+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.05259","created_at":"2026-07-05T10:45:36.442437+00:00"},{"alias_kind":"pith_short_12","alias_value":"3EYWV6OGQQKR","created_at":"2026-07-05T10:45:36.442437+00:00"},{"alias_kind":"pith_short_16","alias_value":"3EYWV6OGQQKRZM7U","created_at":"2026-07-05T10:45:36.442437+00:00"},{"alias_kind":"pith_short_8","alias_value":"3EYWV6OG","created_at":"2026-07-05T10:45:36.442437+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.15377","citing_title":"Ensemble Monitoring for AI Control: Diverse Signals Outweigh More Compute","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2602.17753","citing_title":"The 2025 AI Agent Index: Documenting Technical and Safety Features of Deployed Agentic AI Systems","ref_index":73,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08321","citing_title":"LLM Wardens: Mitigating Adversarial Persuasion with Third-Party Conversational Oversight","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15384","citing_title":"LinuxArena: A Control Setting for AI Agents in Live Production Software Environments","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17663","citing_title":"ATLAS: Constitution-Conditioned Latent Geometry and Redistribution Across Language Models and Neural Perturbation Data","ref_index":32,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3EYWV6OGQQKRZM7UVQAZNVMOUG","json":"https://pith.science/pith/3EYWV6OGQQKRZM7UVQAZNVMOUG.json","graph_json":"https://pith.science/api/pith-number/3EYWV6OGQQKRZM7UVQAZNVMOUG/graph.json","events_json":"https://pith.science/api/pith-number/3EYWV6OGQQKRZM7UVQAZNVMOUG/events.json","paper":"https://pith.science/paper/3EYWV6OG"},"agent_actions":{"view_html":"https://pith.science/pith/3EYWV6OGQQKRZM7UVQAZNVMOUG","download_json":"https://pith.science/pith/3EYWV6OGQQKRZM7UVQAZNVMOUG.json","view_paper":"https://pith.science/paper/3EYWV6OG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.05259&json=true","fetch_graph":"https://pith.science/api/pith-number/3EYWV6OGQQKRZM7UVQAZNVMOUG/graph.json","fetch_events":"https://pith.science/api/pith-number/3EYWV6OGQQKRZM7UVQAZNVMOUG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3EYWV6OGQQKRZM7UVQAZNVMOUG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3EYWV6OGQQKRZM7UVQAZNVMOUG/action/storage_attestation","attest_author":"https://pith.science/pith/3EYWV6OGQQKRZM7UVQAZNVMOUG/action/author_attestation","sign_citation":"https://pith.science/pith/3EYWV6OGQQKRZM7UVQAZNVMOUG/action/citation_signature","submit_replication":"https://pith.science/pith/3EYWV6OGQQKRZM7UVQAZNVMOUG/action/replication_record"}},"created_at":"2026-07-05T10:45:36.442437+00:00","updated_at":"2026-07-05T10:45:36.442437+00:00"}