{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:2Y2RPHVWTDF6XXZUYEM6JJXIEE","short_pith_number":"pith:2Y2RPHVW","schema_version":"1.0","canonical_sha256":"d635179eb698cbebdf34c119e4a6e8210d224f36f3034f25a8310e16007967b5","source":{"kind":"arxiv","id":"2507.00239","version":1},"attestation_state":"computed","paper":{"title":"Linearly Decoding Refused Knowledge in Aligned Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Ari Holtzman, Aryan Shrivastava","submitted_at":"2025-06-30T20:13:49Z","abstract_excerpt":"Most commonly used language models (LMs) are instruction-tuned and aligned using a combination of fine-tuning and reinforcement learning, causing them to refuse users requests deemed harmful by the model. However, jailbreak prompts can often bypass these refusal mechanisms and elicit harmful responses. In this work, we study the extent to which information accessed via jailbreak prompts is decodable using linear probes trained on LM hidden states. We show that a great deal of initially refused information is linearly decodable. For example, across models, the response of a jailbroken LM for th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.00239","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-30T20:13:49Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"d709069de16153d214a3fc007ae559730df880ac102a661462ac1bc49b51401a","abstract_canon_sha256":"7c802a3ec6efccc66c580d3392579a3bf9bb0f7f6c11a41f85ef8b2a5d38b0cb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:30:04.940726Z","signature_b64":"jkfgO1P42kFKOYuQ7rry1IQBAPgHMRzb0w2K9sPdKOiOf3QMDCICiIlZOGbIjQ0jCUnkGli8zL7N3kS4Iq1HCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d635179eb698cbebdf34c119e4a6e8210d224f36f3034f25a8310e16007967b5","last_reissued_at":"2026-07-05T11:30:04.940235Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:30:04.940235Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Linearly Decoding Refused Knowledge in Aligned Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Ari Holtzman, Aryan Shrivastava","submitted_at":"2025-06-30T20:13:49Z","abstract_excerpt":"Most commonly used language models (LMs) are instruction-tuned and aligned using a combination of fine-tuning and reinforcement learning, causing them to refuse users requests deemed harmful by the model. However, jailbreak prompts can often bypass these refusal mechanisms and elicit harmful responses. In this work, we study the extent to which information accessed via jailbreak prompts is decodable using linear probes trained on LM hidden states. We show that a great deal of initially refused information is linearly decodable. For example, across models, the response of a jailbroken LM for th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.00239","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.00239/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.00239","created_at":"2026-07-05T11:30:04.940292+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.00239v1","created_at":"2026-07-05T11:30:04.940292+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.00239","created_at":"2026-07-05T11:30:04.940292+00:00"},{"alias_kind":"pith_short_12","alias_value":"2Y2RPHVWTDF6","created_at":"2026-07-05T11:30:04.940292+00:00"},{"alias_kind":"pith_short_16","alias_value":"2Y2RPHVWTDF6XXZU","created_at":"2026-07-05T11:30:04.940292+00:00"},{"alias_kind":"pith_short_8","alias_value":"2Y2RPHVW","created_at":"2026-07-05T11:30:04.940292+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.19341","citing_title":"HalluWorld: A Controlled Benchmark for Hallucination via Reference World Models","ref_index":57,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2Y2RPHVWTDF6XXZUYEM6JJXIEE","json":"https://pith.science/pith/2Y2RPHVWTDF6XXZUYEM6JJXIEE.json","graph_json":"https://pith.science/api/pith-number/2Y2RPHVWTDF6XXZUYEM6JJXIEE/graph.json","events_json":"https://pith.science/api/pith-number/2Y2RPHVWTDF6XXZUYEM6JJXIEE/events.json","paper":"https://pith.science/paper/2Y2RPHVW"},"agent_actions":{"view_html":"https://pith.science/pith/2Y2RPHVWTDF6XXZUYEM6JJXIEE","download_json":"https://pith.science/pith/2Y2RPHVWTDF6XXZUYEM6JJXIEE.json","view_paper":"https://pith.science/paper/2Y2RPHVW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.00239&json=true","fetch_graph":"https://pith.science/api/pith-number/2Y2RPHVWTDF6XXZUYEM6JJXIEE/graph.json","fetch_events":"https://pith.science/api/pith-number/2Y2RPHVWTDF6XXZUYEM6JJXIEE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2Y2RPHVWTDF6XXZUYEM6JJXIEE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2Y2RPHVWTDF6XXZUYEM6JJXIEE/action/storage_attestation","attest_author":"https://pith.science/pith/2Y2RPHVWTDF6XXZUYEM6JJXIEE/action/author_attestation","sign_citation":"https://pith.science/pith/2Y2RPHVWTDF6XXZUYEM6JJXIEE/action/citation_signature","submit_replication":"https://pith.science/pith/2Y2RPHVWTDF6XXZUYEM6JJXIEE/action/replication_record"}},"created_at":"2026-07-05T11:30:04.940292+00:00","updated_at":"2026-07-05T11:30:04.940292+00:00"}