{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:N3N2JGNGE7LMQ3ULZKXQLU5P6M","short_pith_number":"pith:N3N2JGNG","schema_version":"1.0","canonical_sha256":"6edba499a627d6c86e8bcaaf05d3aff3329224ccca29f5b9575001f72f4ab97f","source":{"kind":"arxiv","id":"2412.09565","version":2},"attestation_state":"computed","paper":{"title":"Obfuscated Activations Bypass LLM Latent-Space Defenses","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Abhay Sheshadri, Alex Serrano, Carlos Guestrin, Erik Jenner, Jacob Hilton, Jordan Taylor, Luke Bailey, Mikhail Seleznyov, Scott Emmons, Stephen Casper","submitted_at":"2024-12-12T18:49:53Z","abstract_excerpt":"Recent latent-space monitoring techniques have shown promise as defenses against LLM attacks. These defenses act as scanners that seek to detect harmful activations before they lead to undesirable actions. This prompts the question: Can models execute harmful behavior via inconspicuous latent states? Here, we study such obfuscated activations. We show that state-of-the-art latent-space defenses -- including sparse autoencoders, representation probing, and latent OOD detection -- are all vulnerable to obfuscated activations. For example, against probes trained to classify harmfulness, our attac"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.09565","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-12-12T18:49:53Z","cross_cats_sorted":[],"title_canon_sha256":"ed88e00a7ea48a6d3fca70b92d525803b4f471f24fb4814de080bc1d059825a9","abstract_canon_sha256":"6a21072b7986e0cdd14ee02e4e03a8fba39a19ba756720658d1f567c63d2622f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:11:30.206001Z","signature_b64":"4d77jdGD4YUThRVa5BJwkWFKiiVabu7hEnwI1D3bOkpMLdfRumH97bUyHrP6bWs2OILKgYXVXkJpwSe6SYYeCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6edba499a627d6c86e8bcaaf05d3aff3329224ccca29f5b9575001f72f4ab97f","last_reissued_at":"2026-07-05T10:11:30.205421Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:11:30.205421Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Obfuscated Activations Bypass LLM Latent-Space Defenses","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Abhay Sheshadri, Alex Serrano, Carlos Guestrin, Erik Jenner, Jacob Hilton, Jordan Taylor, Luke Bailey, Mikhail Seleznyov, Scott Emmons, Stephen Casper","submitted_at":"2024-12-12T18:49:53Z","abstract_excerpt":"Recent latent-space monitoring techniques have shown promise as defenses against LLM attacks. These defenses act as scanners that seek to detect harmful activations before they lead to undesirable actions. This prompts the question: Can models execute harmful behavior via inconspicuous latent states? Here, we study such obfuscated activations. We show that state-of-the-art latent-space defenses -- including sparse autoencoders, representation probing, and latent OOD detection -- are all vulnerable to obfuscated activations. For example, against probes trained to classify harmfulness, our attac"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.09565","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.09565/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.09565","created_at":"2026-07-05T10:11:30.205495+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.09565v2","created_at":"2026-07-05T10:11:30.205495+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.09565","created_at":"2026-07-05T10:11:30.205495+00:00"},{"alias_kind":"pith_short_12","alias_value":"N3N2JGNGE7LM","created_at":"2026-07-05T10:11:30.205495+00:00"},{"alias_kind":"pith_short_16","alias_value":"N3N2JGNGE7LMQ3UL","created_at":"2026-07-05T10:11:30.205495+00:00"},{"alias_kind":"pith_short_8","alias_value":"N3N2JGNG","created_at":"2026-07-05T10:11:30.205495+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12360","citing_title":"Anatomy of Post-Training: Using Interpretability to Characterize Data and Shape the Learning Signal","ref_index":170,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18284","citing_title":"Breaking the Solver Bottleneck: Training Task Generators at the Learnable Frontier","ref_index":104,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09563","citing_title":"PRISM: Recovering Instruction Sets from Language Model Activations","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08044","citing_title":"When Behavioral Safety Evaluation Fails: A Representation-Level Perspective","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27958","citing_title":"Pressure-Testing Deception Probes in LLMs: Scaling, Robustness, and the Geometry of Deceptive Representations","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2506.06414","citing_title":"Benchmarking Misuse Mitigation Against Covert Adversaries","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12813","citing_title":"REALISTA: Realistic Latent Adversarial Attacks that Elicit LLM Hallucinations","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00269","citing_title":"How Language Models Process Out-of-Distribution Inputs: A Two-Pathway Framework","ref_index":56,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/N3N2JGNGE7LMQ3ULZKXQLU5P6M","json":"https://pith.science/pith/N3N2JGNGE7LMQ3ULZKXQLU5P6M.json","graph_json":"https://pith.science/api/pith-number/N3N2JGNGE7LMQ3ULZKXQLU5P6M/graph.json","events_json":"https://pith.science/api/pith-number/N3N2JGNGE7LMQ3ULZKXQLU5P6M/events.json","paper":"https://pith.science/paper/N3N2JGNG"},"agent_actions":{"view_html":"https://pith.science/pith/N3N2JGNGE7LMQ3ULZKXQLU5P6M","download_json":"https://pith.science/pith/N3N2JGNGE7LMQ3ULZKXQLU5P6M.json","view_paper":"https://pith.science/paper/N3N2JGNG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.09565&json=true","fetch_graph":"https://pith.science/api/pith-number/N3N2JGNGE7LMQ3ULZKXQLU5P6M/graph.json","fetch_events":"https://pith.science/api/pith-number/N3N2JGNGE7LMQ3ULZKXQLU5P6M/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/N3N2JGNGE7LMQ3ULZKXQLU5P6M/action/timestamp_anchor","attest_storage":"https://pith.science/pith/N3N2JGNGE7LMQ3ULZKXQLU5P6M/action/storage_attestation","attest_author":"https://pith.science/pith/N3N2JGNGE7LMQ3ULZKXQLU5P6M/action/author_attestation","sign_citation":"https://pith.science/pith/N3N2JGNGE7LMQ3ULZKXQLU5P6M/action/citation_signature","submit_replication":"https://pith.science/pith/N3N2JGNGE7LMQ3ULZKXQLU5P6M/action/replication_record"}},"created_at":"2026-07-05T10:11:30.205495+00:00","updated_at":"2026-07-05T10:11:30.205495+00:00"}