{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:U7IP2S6ICPOCWRINIEUEJMC7RF","short_pith_number":"pith:U7IP2S6I","schema_version":"1.0","canonical_sha256":"a7d0fd4bc813dc2b450d412844b05f894ba116f5827e797099168acf4ca5435a","source":{"kind":"arxiv","id":"2412.12480","version":4},"attestation_state":"computed","paper":{"title":"Subversion Strategy Eval: Can language models statelessly strategize to subvert control protocols?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Alessandro Abate, Alex Mallen, Buck Shlegeris, Charlie Griffin, Misha Wagner","submitted_at":"2024-12-17T02:33:45Z","abstract_excerpt":"An AI control protocol is a plan for usefully deploying AI systems that aims to prevent an AI from intentionally causing some unacceptable outcome. This paper investigates how well AI systems can generate and act on their own strategies for subverting control protocols whilst operating statelessly (without shared memory between contexts). To do this, an AI system may need to reliably generate optimal plans in each context, take actions with well-calibrated probabilities, and coordinate plans with other instances of itself without communicating. We develop Subversion Strategy Eval, a suite of e"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.12480","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-12-17T02:33:45Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"36fb25ca629b2af1a64c410a94f48371e3ace8b61da7ce6148aecbbebd6bfb6a","abstract_canon_sha256":"a50b2a3c6f71e53b44a28f61b9c43c65afeed31287aa0174de5bb6ac2a536571"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:44:20.864098Z","signature_b64":"YF0xClHzeXnBoPTPxFE+/rVfGedDRRbw2eGZDvTe6QfX8GczT0qeQJBXwNcyxJGKEQW2dhGuoROBmIyveWLBBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a7d0fd4bc813dc2b450d412844b05f894ba116f5827e797099168acf4ca5435a","last_reissued_at":"2026-07-05T10:44:20.863647Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:44:20.863647Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Subversion Strategy Eval: Can language models statelessly strategize to subvert control protocols?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Alessandro Abate, Alex Mallen, Buck Shlegeris, Charlie Griffin, Misha Wagner","submitted_at":"2024-12-17T02:33:45Z","abstract_excerpt":"An AI control protocol is a plan for usefully deploying AI systems that aims to prevent an AI from intentionally causing some unacceptable outcome. This paper investigates how well AI systems can generate and act on their own strategies for subverting control protocols whilst operating statelessly (without shared memory between contexts). To do this, an AI system may need to reliably generate optimal plans in each context, take actions with well-calibrated probabilities, and coordinate plans with other instances of itself without communicating. We develop Subversion Strategy Eval, a suite of e"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.12480","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.12480/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.12480","created_at":"2026-07-05T10:44:20.863703+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.12480v4","created_at":"2026-07-05T10:44:20.863703+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.12480","created_at":"2026-07-05T10:44:20.863703+00:00"},{"alias_kind":"pith_short_12","alias_value":"U7IP2S6ICPOC","created_at":"2026-07-05T10:44:20.863703+00:00"},{"alias_kind":"pith_short_16","alias_value":"U7IP2S6ICPOCWRIN","created_at":"2026-07-05T10:44:20.863703+00:00"},{"alias_kind":"pith_short_8","alias_value":"U7IP2S6I","created_at":"2026-07-05T10:44:20.863703+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.06529","citing_title":"Attack Selection in Agentic AI Control Evaluations Meaningfully Decreases Safety","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15377","citing_title":"Ensemble Monitoring for AI Control: Diverse Signals Outweigh More Compute","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15384","citing_title":"LinuxArena: A Control Setting for AI Agents in Live Production Software Environments","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/U7IP2S6ICPOCWRINIEUEJMC7RF","json":"https://pith.science/pith/U7IP2S6ICPOCWRINIEUEJMC7RF.json","graph_json":"https://pith.science/api/pith-number/U7IP2S6ICPOCWRINIEUEJMC7RF/graph.json","events_json":"https://pith.science/api/pith-number/U7IP2S6ICPOCWRINIEUEJMC7RF/events.json","paper":"https://pith.science/paper/U7IP2S6I"},"agent_actions":{"view_html":"https://pith.science/pith/U7IP2S6ICPOCWRINIEUEJMC7RF","download_json":"https://pith.science/pith/U7IP2S6ICPOCWRINIEUEJMC7RF.json","view_paper":"https://pith.science/paper/U7IP2S6I","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.12480&json=true","fetch_graph":"https://pith.science/api/pith-number/U7IP2S6ICPOCWRINIEUEJMC7RF/graph.json","fetch_events":"https://pith.science/api/pith-number/U7IP2S6ICPOCWRINIEUEJMC7RF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/U7IP2S6ICPOCWRINIEUEJMC7RF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/U7IP2S6ICPOCWRINIEUEJMC7RF/action/storage_attestation","attest_author":"https://pith.science/pith/U7IP2S6ICPOCWRINIEUEJMC7RF/action/author_attestation","sign_citation":"https://pith.science/pith/U7IP2S6ICPOCWRINIEUEJMC7RF/action/citation_signature","submit_replication":"https://pith.science/pith/U7IP2S6ICPOCWRINIEUEJMC7RF/action/replication_record"}},"created_at":"2026-07-05T10:44:20.863703+00:00","updated_at":"2026-07-05T10:44:20.863703+00:00"}