{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:2Y2RPHVWTDF6XXZUYEM6JJXIEE","short_pith_number":"pith:2Y2RPHVW","canonical_record":{"source":{"id":"2507.00239","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-30T20:13:49Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"d709069de16153d214a3fc007ae559730df880ac102a661462ac1bc49b51401a","abstract_canon_sha256":"7c802a3ec6efccc66c580d3392579a3bf9bb0f7f6c11a41f85ef8b2a5d38b0cb"},"schema_version":"1.0"},"canonical_sha256":"d635179eb698cbebdf34c119e4a6e8210d224f36f3034f25a8310e16007967b5","source":{"kind":"arxiv","id":"2507.00239","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2507.00239","created_at":"2026-07-05T11:30:04Z"},{"alias_kind":"arxiv_version","alias_value":"2507.00239v1","created_at":"2026-07-05T11:30:04Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.00239","created_at":"2026-07-05T11:30:04Z"},{"alias_kind":"pith_short_12","alias_value":"2Y2RPHVWTDF6","created_at":"2026-07-05T11:30:04Z"},{"alias_kind":"pith_short_16","alias_value":"2Y2RPHVWTDF6XXZU","created_at":"2026-07-05T11:30:04Z"},{"alias_kind":"pith_short_8","alias_value":"2Y2RPHVW","created_at":"2026-07-05T11:30:04Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:2Y2RPHVWTDF6XXZUYEM6JJXIEE","target":"record","payload":{"canonical_record":{"source":{"id":"2507.00239","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-30T20:13:49Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"d709069de16153d214a3fc007ae559730df880ac102a661462ac1bc49b51401a","abstract_canon_sha256":"7c802a3ec6efccc66c580d3392579a3bf9bb0f7f6c11a41f85ef8b2a5d38b0cb"},"schema_version":"1.0"},"canonical_sha256":"d635179eb698cbebdf34c119e4a6e8210d224f36f3034f25a8310e16007967b5","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:30:04.940726Z","signature_b64":"jkfgO1P42kFKOYuQ7rry1IQBAPgHMRzb0w2K9sPdKOiOf3QMDCICiIlZOGbIjQ0jCUnkGli8zL7N3kS4Iq1HCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d635179eb698cbebdf34c119e4a6e8210d224f36f3034f25a8310e16007967b5","last_reissued_at":"2026-07-05T11:30:04.940235Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:30:04.940235Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2507.00239","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:30:04Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"D27t+zQ8N7xuEhlEDpFDWgoTtM+eiQyb3xXlRCTh4o6I7HRsnHWOOAGOlh2nMaeHISKxwJlF2zHUNY/PfedfCg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T21:56:02.680763Z"},"content_sha256":"2db72ace813a6983e2d8390628a71243667cc9eb55dc821e980a78fd7fe1ae4f","schema_version":"1.0","event_id":"sha256:2db72ace813a6983e2d8390628a71243667cc9eb55dc821e980a78fd7fe1ae4f"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:2Y2RPHVWTDF6XXZUYEM6JJXIEE","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Linearly Decoding Refused Knowledge in Aligned Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Ari Holtzman, Aryan Shrivastava","submitted_at":"2025-06-30T20:13:49Z","abstract_excerpt":"Most commonly used language models (LMs) are instruction-tuned and aligned using a combination of fine-tuning and reinforcement learning, causing them to refuse users requests deemed harmful by the model. However, jailbreak prompts can often bypass these refusal mechanisms and elicit harmful responses. In this work, we study the extent to which information accessed via jailbreak prompts is decodable using linear probes trained on LM hidden states. We show that a great deal of initially refused information is linearly decodable. For example, across models, the response of a jailbroken LM for th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.00239","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.00239/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:30:04Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"1MeIcOjGSK8vDqRCkgWocGgdd7ePDlq/tH4HkKp8+EcBaQhhJybQAGOm5jeO4M35N7PBTOzkq6Dl0jmE4dBXDw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T21:56:02.681264Z"},"content_sha256":"5c6c7fb335ba7959146b5a55a025ad66eb4a0b894ba0a4bdf5be38fc118e4d25","schema_version":"1.0","event_id":"sha256:5c6c7fb335ba7959146b5a55a025ad66eb4a0b894ba0a4bdf5be38fc118e4d25"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/2Y2RPHVWTDF6XXZUYEM6JJXIEE/bundle.json","state_url":"https://pith.science/pith/2Y2RPHVWTDF6XXZUYEM6JJXIEE/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/2Y2RPHVWTDF6XXZUYEM6JJXIEE/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-09T21:56:02Z","links":{"resolver":"https://pith.science/pith/2Y2RPHVWTDF6XXZUYEM6JJXIEE","bundle":"https://pith.science/pith/2Y2RPHVWTDF6XXZUYEM6JJXIEE/bundle.json","state":"https://pith.science/pith/2Y2RPHVWTDF6XXZUYEM6JJXIEE/state.json","well_known_bundle":"https://pith.science/.well-known/pith/2Y2RPHVWTDF6XXZUYEM6JJXIEE/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:2Y2RPHVWTDF6XXZUYEM6JJXIEE","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"7c802a3ec6efccc66c580d3392579a3bf9bb0f7f6c11a41f85ef8b2a5d38b0cb","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-30T20:13:49Z","title_canon_sha256":"d709069de16153d214a3fc007ae559730df880ac102a661462ac1bc49b51401a"},"schema_version":"1.0","source":{"id":"2507.00239","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2507.00239","created_at":"2026-07-05T11:30:04Z"},{"alias_kind":"arxiv_version","alias_value":"2507.00239v1","created_at":"2026-07-05T11:30:04Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.00239","created_at":"2026-07-05T11:30:04Z"},{"alias_kind":"pith_short_12","alias_value":"2Y2RPHVWTDF6","created_at":"2026-07-05T11:30:04Z"},{"alias_kind":"pith_short_16","alias_value":"2Y2RPHVWTDF6XXZU","created_at":"2026-07-05T11:30:04Z"},{"alias_kind":"pith_short_8","alias_value":"2Y2RPHVW","created_at":"2026-07-05T11:30:04Z"}],"graph_snapshots":[{"event_id":"sha256:5c6c7fb335ba7959146b5a55a025ad66eb4a0b894ba0a4bdf5be38fc118e4d25","target":"graph","created_at":"2026-07-05T11:30:04Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2507.00239/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Most commonly used language models (LMs) are instruction-tuned and aligned using a combination of fine-tuning and reinforcement learning, causing them to refuse users requests deemed harmful by the model. However, jailbreak prompts can often bypass these refusal mechanisms and elicit harmful responses. In this work, we study the extent to which information accessed via jailbreak prompts is decodable using linear probes trained on LM hidden states. We show that a great deal of initially refused information is linearly decodable. For example, across models, the response of a jailbroken LM for th","authors_text":"Ari Holtzman, Aryan Shrivastava","cross_cats":["cs.AI"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-30T20:13:49Z","title":"Linearly Decoding Refused Knowledge in Aligned Language Models"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.00239","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:2db72ace813a6983e2d8390628a71243667cc9eb55dc821e980a78fd7fe1ae4f","target":"record","created_at":"2026-07-05T11:30:04Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"7c802a3ec6efccc66c580d3392579a3bf9bb0f7f6c11a41f85ef8b2a5d38b0cb","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-30T20:13:49Z","title_canon_sha256":"d709069de16153d214a3fc007ae559730df880ac102a661462ac1bc49b51401a"},"schema_version":"1.0","source":{"id":"2507.00239","kind":"arxiv","version":1}},"canonical_sha256":"d635179eb698cbebdf34c119e4a6e8210d224f36f3034f25a8310e16007967b5","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"d635179eb698cbebdf34c119e4a6e8210d224f36f3034f25a8310e16007967b5","first_computed_at":"2026-07-05T11:30:04.940235Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:30:04.940235Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"jkfgO1P42kFKOYuQ7rry1IQBAPgHMRzb0w2K9sPdKOiOf3QMDCICiIlZOGbIjQ0jCUnkGli8zL7N3kS4Iq1HCw==","signature_status":"signed_v1","signed_at":"2026-07-05T11:30:04.940726Z","signed_message":"canonical_sha256_bytes"},"source_id":"2507.00239","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:2db72ace813a6983e2d8390628a71243667cc9eb55dc821e980a78fd7fe1ae4f","sha256:5c6c7fb335ba7959146b5a55a025ad66eb4a0b894ba0a4bdf5be38fc118e4d25"],"state_sha256":"5fe4ee015face443beb973ef14ac95895f6962a3b0eb326e0675fb9425e29482"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"PyXNdsoVcxRLcbBgpY+W1Sws5OszNc1ktONzHN+/kxJs01QpVjOIHBvFZX/k5750TWLM7iptRoozZUQPDsduDw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-09T21:56:02.686613Z","bundle_sha256":"8edef7c69e58dc3bdaec9cab2635cbf06059dcee18ee1e3a83e84017a5978fce"}}