{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:IXH56MIJEPLGDXQUVQNTN36MZY","short_pith_number":"pith:IXH56MIJ","schema_version":"1.0","canonical_sha256":"45cfdf310923d661de14ac1b36efccce3674b84c7adc7501cf241e2d32268356","source":{"kind":"arxiv","id":"2312.06942","version":5},"attestation_state":"computed","paper":{"title":"AI Control: Improving Safety Despite Intentional Subversion","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Buck Shlegeris, Fabien Roger, Kshitij Sachan, Ryan Greenblatt","submitted_at":"2023-12-12T02:34:06Z","abstract_excerpt":"As large language models (LLMs) become more powerful and are deployed more autonomously, it will be increasingly important to prevent them from causing harmful outcomes. Researchers have investigated a variety of safety techniques for this purpose, e.g. using models to review the outputs of other models, or red-teaming techniques to surface subtle failure modes. However, researchers have not evaluated whether such techniques still ensure safety if the model is itself intentionally trying to subvert them. In this paper, we develop and evaluate pipelines of safety techniques (\"protocols\") that a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.06942","kind":"arxiv","version":5},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-12-12T02:34:06Z","cross_cats_sorted":[],"title_canon_sha256":"a52a2846fb875879ad6a71f0acf4df511643a9a9f06b58dde6ba44a5c96c93a3","abstract_canon_sha256":"dc69141ab974b8dae420924735e9c63de179e7f5811c50f85d142f9a9a9d1655"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:47:09.252374Z","signature_b64":"zYCtR+h7FFi6leKIf+VyiZ/B3wfIy6YJYXp3BAPi40g1yaA2PQKej+Xb3apBjxCrPqNxkdxsgGxsovp6z/m4BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"45cfdf310923d661de14ac1b36efccce3674b84c7adc7501cf241e2d32268356","last_reissued_at":"2026-07-05T08:47:09.251835Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:47:09.251835Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AI Control: Improving Safety Despite Intentional Subversion","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Buck Shlegeris, Fabien Roger, Kshitij Sachan, Ryan Greenblatt","submitted_at":"2023-12-12T02:34:06Z","abstract_excerpt":"As large language models (LLMs) become more powerful and are deployed more autonomously, it will be increasingly important to prevent them from causing harmful outcomes. Researchers have investigated a variety of safety techniques for this purpose, e.g. using models to review the outputs of other models, or red-teaming techniques to surface subtle failure modes. However, researchers have not evaluated whether such techniques still ensure safety if the model is itself intentionally trying to subvert them. In this paper, we develop and evaluate pipelines of safety techniques (\"protocols\") that a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.06942","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.06942/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.06942","created_at":"2026-07-05T08:47:09.251906+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.06942v5","created_at":"2026-07-05T08:47:09.251906+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.06942","created_at":"2026-07-05T08:47:09.251906+00:00"},{"alias_kind":"pith_short_12","alias_value":"IXH56MIJEPLG","created_at":"2026-07-05T08:47:09.251906+00:00"},{"alias_kind":"pith_short_16","alias_value":"IXH56MIJEPLGDXQU","created_at":"2026-07-05T08:47:09.251906+00:00"},{"alias_kind":"pith_short_8","alias_value":"IXH56MIJ","created_at":"2026-07-05T08:47:09.251906+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":24,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2607.07774","citing_title":"ScopeJudge: Cost-Aware Pre-Execution Gating for Offensive Security Agents","ref_index":5,"is_internal_anchor":true},{"citing_arxiv_id":"2607.08066","citing_title":"Persuasion Attacks Can Decrease Effectiveness of CoT Monitoring","ref_index":87,"is_internal_anchor":true},{"citing_arxiv_id":"2605.23935","citing_title":"Operationalizing Reconstructive Authority: Runtime Construction, Dependency Resolution, and Execution Gating in Autonomous Agent Systems","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00036","citing_title":"AI Integrity: Defending Against Backdoors and Secret Loyalties","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02510","citing_title":"Online Safety Monitoring for LLMs","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12618","citing_title":"\"Did you lie?\" Evaluating Lie Detectors across Model Scale and Belief-Verified Model Organisms","ref_index":72,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10456","citing_title":"The Distributed Detectability Band Against Marginal-Preserving Attacks","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07054","citing_title":"TRACE: Trajectory Reasoning through Adaptive Cross-Step Evidence Aggregation for LLM Agents","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05330","citing_title":"A Model of Multi-turn Human Persuadability Using Probabilistic Belief Tracing","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02837","citing_title":"Fixing FOLIO and MALLS: Verified Annotations and an LLM-assisted Framework to Focus Human Relabeling","ref_index":92,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28425","citing_title":"Tool Use Enables Undetectable Steganography in Multi-Agent LLM Systems","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05194","citing_title":"Temporal Preference Concepts and their Functions in a Large Language Model","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24286","citing_title":"Faithfulness as Information Flow: Evaluating and Training Faithful Chain-of-Thought Reasoning","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22643","citing_title":"Boiling the Frog: A Multi-Turn Benchmark for Agentic Safety","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22643","citing_title":"Boiling the Frog: A Multi-Turn Benchmark for Agentic Safety","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2506.02546","citing_title":"To trust or not to trust: Attention-based Trust Management for LLM Multi-Agent Systems","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13069","citing_title":"Geographic Blind Spots in AI Control Monitors: A Cross-National Audit of Claude Opus 4.6","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.01151","citing_title":"Detecting Multi-Agent Collusion Through Multi-Agent Interpretability","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24966","citing_title":"Risk Reporting for Developers' Internal AI Model Use","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06390","citing_title":"Automated alignment is harder than you think","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22167","citing_title":"Estimating Tail Risks in Language Model Output Distributions","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11806","citing_title":"Detecting Safety Violations Across Many Agent Traces","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17517","citing_title":"From Admission to Invariants: Measuring Deviation in Delegated Agent Systems","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17663","citing_title":"ATLAS: Constitution-Conditioned Latent Geometry and Redistribution Across Language Models and Neural Perturbation Data","ref_index":30,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IXH56MIJEPLGDXQUVQNTN36MZY","json":"https://pith.science/pith/IXH56MIJEPLGDXQUVQNTN36MZY.json","graph_json":"https://pith.science/api/pith-number/IXH56MIJEPLGDXQUVQNTN36MZY/graph.json","events_json":"https://pith.science/api/pith-number/IXH56MIJEPLGDXQUVQNTN36MZY/events.json","paper":"https://pith.science/paper/IXH56MIJ"},"agent_actions":{"view_html":"https://pith.science/pith/IXH56MIJEPLGDXQUVQNTN36MZY","download_json":"https://pith.science/pith/IXH56MIJEPLGDXQUVQNTN36MZY.json","view_paper":"https://pith.science/paper/IXH56MIJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.06942&json=true","fetch_graph":"https://pith.science/api/pith-number/IXH56MIJEPLGDXQUVQNTN36MZY/graph.json","fetch_events":"https://pith.science/api/pith-number/IXH56MIJEPLGDXQUVQNTN36MZY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IXH56MIJEPLGDXQUVQNTN36MZY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IXH56MIJEPLGDXQUVQNTN36MZY/action/storage_attestation","attest_author":"https://pith.science/pith/IXH56MIJEPLGDXQUVQNTN36MZY/action/author_attestation","sign_citation":"https://pith.science/pith/IXH56MIJEPLGDXQUVQNTN36MZY/action/citation_signature","submit_replication":"https://pith.science/pith/IXH56MIJEPLGDXQUVQNTN36MZY/action/replication_record"}},"created_at":"2026-07-05T08:47:09.251906+00:00","updated_at":"2026-07-05T08:47:09.251906+00:00"}