{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:URMW7OATDXGQEQCYXI2WZDOTW6","short_pith_number":"pith:URMW7OAT","schema_version":"1.0","canonical_sha256":"a4596fb8131dcd024058ba356c8dd3b79cc54da04b2a5bb13ae997e35a596036","source":{"kind":"arxiv","id":"2410.21514","version":1},"attestation_state":"computed","paper":{"title":"Sabotage Evaluations for Frontier Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CY"],"primary_cat":"cs.LG","authors_text":"Buck Shlegeris, Cem Anil, David Duvenaud, Deep Ganguli, Eric Christiansen, Esin Durmus, Ethan Perez, Evan Hubinger, Holden Karnofsky, Jai Srivastav, Jared Kaplan, Joe Benton, Misha Wagner, Roger Grosse, Samuel R. Bowman, Shauna Kravec","submitted_at":"2024-10-28T20:34:51Z","abstract_excerpt":"Sufficiently capable models could subvert human oversight and decision-making in important contexts. For example, in the context of AI development, models could covertly sabotage efforts to evaluate their own dangerous capabilities, to monitor their behavior, or to make decisions about their deployment. We refer to this family of abilities as sabotage capabilities. We develop a set of related threat models and evaluations. These evaluations are designed to provide evidence that a given model, operating under a given set of mitigations, could not successfully sabotage a frontier model developer"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.21514","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-10-28T20:34:51Z","cross_cats_sorted":["cs.AI","cs.CY"],"title_canon_sha256":"d6e7f142cc97eb74cb2f6111972764534052ee6be81abdb9678a396876304fa3","abstract_canon_sha256":"c851fa598fcb8e3bbd35fe7733e140c0e623d678f64c49aaee225c853e44f211"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:27:29.324609Z","signature_b64":"eQe53f2h9KDIDfFn7R1jeWvVxLzVJFOI+Q/GMAJrg+qZAbEkYTMe5hYzFRJsnr1FjgoJmZPO6I3PG4Hb7IkaCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a4596fb8131dcd024058ba356c8dd3b79cc54da04b2a5bb13ae997e35a596036","last_reissued_at":"2026-07-05T09:27:29.324073Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:27:29.324073Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Sabotage Evaluations for Frontier Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CY"],"primary_cat":"cs.LG","authors_text":"Buck Shlegeris, Cem Anil, David Duvenaud, Deep Ganguli, Eric Christiansen, Esin Durmus, Ethan Perez, Evan Hubinger, Holden Karnofsky, Jai Srivastav, Jared Kaplan, Joe Benton, Misha Wagner, Roger Grosse, Samuel R. Bowman, Shauna Kravec","submitted_at":"2024-10-28T20:34:51Z","abstract_excerpt":"Sufficiently capable models could subvert human oversight and decision-making in important contexts. For example, in the context of AI development, models could covertly sabotage efforts to evaluate their own dangerous capabilities, to monitor their behavior, or to make decisions about their deployment. We refer to this family of abilities as sabotage capabilities. We develop a set of related threat models and evaluations. These evaluations are designed to provide evidence that a given model, operating under a given set of mitigations, could not successfully sabotage a frontier model developer"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.21514","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.21514/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.21514","created_at":"2026-07-05T09:27:29.324141+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.21514v1","created_at":"2026-07-05T09:27:29.324141+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.21514","created_at":"2026-07-05T09:27:29.324141+00:00"},{"alias_kind":"pith_short_12","alias_value":"URMW7OATDXGQ","created_at":"2026-07-05T09:27:29.324141+00:00"},{"alias_kind":"pith_short_16","alias_value":"URMW7OATDXGQEQCY","created_at":"2026-07-05T09:27:29.324141+00:00"},{"alias_kind":"pith_short_8","alias_value":"URMW7OAT","created_at":"2026-07-05T09:27:29.324141+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":17,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19603","citing_title":"Comparing Linear Probes with Mahalanobis Cosine Similarity","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17478","citing_title":"Decoding Hidden Deception in Reasoning LLMs: Activation Explainers for Deception Auditing","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12016","citing_title":"Generalization Hacking: Models Can Game Reinforcement Learning by Preventing Behavioral Generalization","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10456","citing_title":"The Distributed Detectability Band Against Marginal-Preserving Attacks","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08629","citing_title":"Sycophancy Towards Researchers Drives Performative Misalignment","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05647","citing_title":"Coding with \"Enemy\": Can Human Developers Detect AI Agent Sabotage?","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06529","citing_title":"Attack Selection in Agentic AI Control Evaluations Meaningfully Decreases Safety","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31916","citing_title":"Theory of Mind and Persuasion Beyond Conversation: Assessing the Capacity of LLMs to Induce Belief States via Planning and Action","ref_index":72,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09391","citing_title":"Do Linear Probes Generalize Better in Persona Coordinates?","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2511.17408","citing_title":"The Impact of Off-Policy Training Data on Probe Generalisation","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03121","citing_title":"An Independent Safety Evaluation of Kimi K2.5","ref_index":85,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12366","citing_title":"Classifier Context Rot: Monitor Performance Degrades with Context Length","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2412.14093","citing_title":"Alignment faking in large language models","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09391","citing_title":"Do Linear Probes Generalize Better in Persona Coordinates?","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05835","citing_title":"Evaluation Awareness in Language Models Has Limited Effect on Behaviour","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13301","citing_title":"Honeypot Protocol","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15384","citing_title":"LinuxArena: A Control Setting for AI Agents in Live Production Software Environments","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/URMW7OATDXGQEQCYXI2WZDOTW6","json":"https://pith.science/pith/URMW7OATDXGQEQCYXI2WZDOTW6.json","graph_json":"https://pith.science/api/pith-number/URMW7OATDXGQEQCYXI2WZDOTW6/graph.json","events_json":"https://pith.science/api/pith-number/URMW7OATDXGQEQCYXI2WZDOTW6/events.json","paper":"https://pith.science/paper/URMW7OAT"},"agent_actions":{"view_html":"https://pith.science/pith/URMW7OATDXGQEQCYXI2WZDOTW6","download_json":"https://pith.science/pith/URMW7OATDXGQEQCYXI2WZDOTW6.json","view_paper":"https://pith.science/paper/URMW7OAT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.21514&json=true","fetch_graph":"https://pith.science/api/pith-number/URMW7OATDXGQEQCYXI2WZDOTW6/graph.json","fetch_events":"https://pith.science/api/pith-number/URMW7OATDXGQEQCYXI2WZDOTW6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/URMW7OATDXGQEQCYXI2WZDOTW6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/URMW7OATDXGQEQCYXI2WZDOTW6/action/storage_attestation","attest_author":"https://pith.science/pith/URMW7OATDXGQEQCYXI2WZDOTW6/action/author_attestation","sign_citation":"https://pith.science/pith/URMW7OATDXGQEQCYXI2WZDOTW6/action/citation_signature","submit_replication":"https://pith.science/pith/URMW7OATDXGQEQCYXI2WZDOTW6/action/replication_record"}},"created_at":"2026-07-05T09:27:29.324141+00:00","updated_at":"2026-07-05T09:27:29.324141+00:00"}