{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:2L6D4PUTDPPYLVNGFOXB3DQXXZ","short_pith_number":"pith:2L6D4PUT","schema_version":"1.0","canonical_sha256":"d2fc3e3e931bdf85d5a62bae1d8e17be5356420abd3ee75cdc62814b7eace0c9","source":{"kind":"arxiv","id":"2406.09289","version":2},"attestation_state":"computed","paper":{"title":"Understanding Jailbreak Success: A Study of Latent Space Dynamics in Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Frauke Kreuter, Nina Panickssery, Sarah Ball","submitted_at":"2024-06-13T16:26:47Z","abstract_excerpt":"Conversational large language models are trained to refuse to answer harmful questions. However, emergent jailbreaking techniques can still elicit unsafe outputs, presenting an ongoing challenge for model alignment. To better understand how different jailbreak types circumvent safeguards, this paper analyses model activations on different jailbreak inputs. We find that it is possible to extract a jailbreak vector from a single class of jailbreaks that works to mitigate jailbreak effectiveness from other semantically-dissimilar classes. This may indicate that different kinds of effective jailbr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.09289","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-06-13T16:26:47Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"d1e93467150738dbb517b95ec865ea9f2c0ece51de0a944260af73ec002d9cb2","abstract_canon_sha256":"019680e740999f5005f7d8991d6c0be92b7dc432ce82ffed027545de3c047d2d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:16:17.989094Z","signature_b64":"B89s+kPEztuwUORB21n0ofU/KQoeFGWPnrhzQTJHbHDo0mWvRKSSB8QpYO0jNFC9QDm5xCDSv8bLV/P2VLWzAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d2fc3e3e931bdf85d5a62bae1d8e17be5356420abd3ee75cdc62814b7eace0c9","last_reissued_at":"2026-07-05T09:16:17.988524Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:16:17.988524Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Understanding Jailbreak Success: A Study of Latent Space Dynamics in Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Frauke Kreuter, Nina Panickssery, Sarah Ball","submitted_at":"2024-06-13T16:26:47Z","abstract_excerpt":"Conversational large language models are trained to refuse to answer harmful questions. However, emergent jailbreaking techniques can still elicit unsafe outputs, presenting an ongoing challenge for model alignment. To better understand how different jailbreak types circumvent safeguards, this paper analyses model activations on different jailbreak inputs. We find that it is possible to extract a jailbreak vector from a single class of jailbreaks that works to mitigate jailbreak effectiveness from other semantically-dissimilar classes. This may indicate that different kinds of effective jailbr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.09289","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.09289/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.09289","created_at":"2026-07-05T09:16:17.988590+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.09289v2","created_at":"2026-07-05T09:16:17.988590+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.09289","created_at":"2026-07-05T09:16:17.988590+00:00"},{"alias_kind":"pith_short_12","alias_value":"2L6D4PUTDPPY","created_at":"2026-07-05T09:16:17.988590+00:00"},{"alias_kind":"pith_short_16","alias_value":"2L6D4PUTDPPYLVNG","created_at":"2026-07-05T09:16:17.988590+00:00"},{"alias_kind":"pith_short_8","alias_value":"2L6D4PUT","created_at":"2026-07-05T09:16:17.988590+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2601.14004","citing_title":"Locate, Steer, and Improve: A Practical Survey of Actionable Mechanistic Interpretability in Large Language Models","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12412","citing_title":"Stories in Space: In-Context Learning Trajectories in Conceptual Belief Space","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11663","citing_title":"Why Do Large Language Models Generate Harmful Content?","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2L6D4PUTDPPYLVNGFOXB3DQXXZ","json":"https://pith.science/pith/2L6D4PUTDPPYLVNGFOXB3DQXXZ.json","graph_json":"https://pith.science/api/pith-number/2L6D4PUTDPPYLVNGFOXB3DQXXZ/graph.json","events_json":"https://pith.science/api/pith-number/2L6D4PUTDPPYLVNGFOXB3DQXXZ/events.json","paper":"https://pith.science/paper/2L6D4PUT"},"agent_actions":{"view_html":"https://pith.science/pith/2L6D4PUTDPPYLVNGFOXB3DQXXZ","download_json":"https://pith.science/pith/2L6D4PUTDPPYLVNGFOXB3DQXXZ.json","view_paper":"https://pith.science/paper/2L6D4PUT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.09289&json=true","fetch_graph":"https://pith.science/api/pith-number/2L6D4PUTDPPYLVNGFOXB3DQXXZ/graph.json","fetch_events":"https://pith.science/api/pith-number/2L6D4PUTDPPYLVNGFOXB3DQXXZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2L6D4PUTDPPYLVNGFOXB3DQXXZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2L6D4PUTDPPYLVNGFOXB3DQXXZ/action/storage_attestation","attest_author":"https://pith.science/pith/2L6D4PUTDPPYLVNGFOXB3DQXXZ/action/author_attestation","sign_citation":"https://pith.science/pith/2L6D4PUTDPPYLVNGFOXB3DQXXZ/action/citation_signature","submit_replication":"https://pith.science/pith/2L6D4PUTDPPYLVNGFOXB3DQXXZ/action/replication_record"}},"created_at":"2026-07-05T09:16:17.988590+00:00","updated_at":"2026-07-05T09:16:17.988590+00:00"}