{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:JPUJOKXJW3RAYJJF6SMPCWHD4U","short_pith_number":"pith:JPUJOKXJ","schema_version":"1.0","canonical_sha256":"4be8972ae9b6e20c2525f498f158e3e50935d3feb155b96886a478479edd78b1","source":{"kind":"arxiv","id":"2504.10374","version":1},"attestation_state":"computed","paper":{"title":"Ctrl-Z: Controlling AI Agents via Resampling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Adam Kaufman, Akbir Khan, Aryan Bhatt, Buck Shlegeris, Cody Rushing, David Matolcsi, Tyler Tracy, Vasil Georgiev","submitted_at":"2025-04-14T16:22:11Z","abstract_excerpt":"Control evaluations measure whether monitoring and security protocols for AI systems prevent intentionally subversive AI models from causing harm. Our work presents the first control evaluation performed in an agent environment. We construct BashBench, a dataset of 257 challenging multi-step system administration tasks, and evaluate whether various safety measures can prevent an adversarially constructed AI agent from covertly downloading and executing malicious code in this environment. This multi-step setting introduces new attack and defense dynamics, which we investigate in order to design"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.10374","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-04-14T16:22:11Z","cross_cats_sorted":[],"title_canon_sha256":"25ecfdb073e566a78f82dcec178f71ea33e06906b7254e59194b5eb0f977190a","abstract_canon_sha256":"15315a734fd427fb3bf7596d368399749ad3e7dfdcd1cc4b4231b5d7e62e34bf"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:48:59.100301Z","signature_b64":"HYBEjyrzynlhQyG29BquhjXYPODD9jB/j2ANnefgoZbe5o/iVeNULSq15Qx8BuLBKk+V+/839rpodgoqeh0tCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4be8972ae9b6e20c2525f498f158e3e50935d3feb155b96886a478479edd78b1","last_reissued_at":"2026-07-05T10:48:59.099831Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:48:59.099831Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Ctrl-Z: Controlling AI Agents via Resampling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Adam Kaufman, Akbir Khan, Aryan Bhatt, Buck Shlegeris, Cody Rushing, David Matolcsi, Tyler Tracy, Vasil Georgiev","submitted_at":"2025-04-14T16:22:11Z","abstract_excerpt":"Control evaluations measure whether monitoring and security protocols for AI systems prevent intentionally subversive AI models from causing harm. Our work presents the first control evaluation performed in an agent environment. We construct BashBench, a dataset of 257 challenging multi-step system administration tasks, and evaluate whether various safety measures can prevent an adversarially constructed AI agent from covertly downloading and executing malicious code in this environment. This multi-step setting introduces new attack and defense dynamics, which we investigate in order to design"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.10374","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.10374/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.10374","created_at":"2026-07-05T10:48:59.099890+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.10374v1","created_at":"2026-07-05T10:48:59.099890+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.10374","created_at":"2026-07-05T10:48:59.099890+00:00"},{"alias_kind":"pith_short_12","alias_value":"JPUJOKXJW3RA","created_at":"2026-07-05T10:48:59.099890+00:00"},{"alias_kind":"pith_short_16","alias_value":"JPUJOKXJW3RAYJJF","created_at":"2026-07-05T10:48:59.099890+00:00"},{"alias_kind":"pith_short_8","alias_value":"JPUJOKXJ","created_at":"2026-07-05T10:48:59.099890+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20663","citing_title":"DrugBench: Evaluating AI Control Protocols for Medication Harm Mitigation","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06529","citing_title":"Attack Selection in Agentic AI Control Evaluations Meaningfully Decreases Safety","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2507.11473","citing_title":"Chain of Thought Monitorability: A New and Fragile Opportunity for AI Safety","ref_index":113,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13301","citing_title":"Honeypot Protocol","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15384","citing_title":"LinuxArena: A Control Setting for AI Agents in Live Production Software Environments","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JPUJOKXJW3RAYJJF6SMPCWHD4U","json":"https://pith.science/pith/JPUJOKXJW3RAYJJF6SMPCWHD4U.json","graph_json":"https://pith.science/api/pith-number/JPUJOKXJW3RAYJJF6SMPCWHD4U/graph.json","events_json":"https://pith.science/api/pith-number/JPUJOKXJW3RAYJJF6SMPCWHD4U/events.json","paper":"https://pith.science/paper/JPUJOKXJ"},"agent_actions":{"view_html":"https://pith.science/pith/JPUJOKXJW3RAYJJF6SMPCWHD4U","download_json":"https://pith.science/pith/JPUJOKXJW3RAYJJF6SMPCWHD4U.json","view_paper":"https://pith.science/paper/JPUJOKXJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.10374&json=true","fetch_graph":"https://pith.science/api/pith-number/JPUJOKXJW3RAYJJF6SMPCWHD4U/graph.json","fetch_events":"https://pith.science/api/pith-number/JPUJOKXJW3RAYJJF6SMPCWHD4U/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JPUJOKXJW3RAYJJF6SMPCWHD4U/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JPUJOKXJW3RAYJJF6SMPCWHD4U/action/storage_attestation","attest_author":"https://pith.science/pith/JPUJOKXJW3RAYJJF6SMPCWHD4U/action/author_attestation","sign_citation":"https://pith.science/pith/JPUJOKXJW3RAYJJF6SMPCWHD4U/action/citation_signature","submit_replication":"https://pith.science/pith/JPUJOKXJW3RAYJJF6SMPCWHD4U/action/replication_record"}},"created_at":"2026-07-05T10:48:59.099890+00:00","updated_at":"2026-07-05T10:48:59.099890+00:00"}