{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ASTQJPYWRNP2OR2BNKI4UANPDV","short_pith_number":"pith:ASTQJPYW","schema_version":"1.0","canonical_sha256":"04a704bf168b5fa747416a91ca01af1d7580822390b255acdc8f710cb35d8125","source":{"kind":"arxiv","id":"2409.00137","version":1},"attestation_state":"computed","paper":{"title":"Emerging Vulnerabilities in Frontier Models: Multi-Turn Jailbreak Attacks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CR","authors_text":"Ethan Kosak-Hine, George Ingebretsen, Jason Zhang, Julius Broomfield, Kellin Pelrine, Reihaneh Iranmanesh, Reihaneh Rabbany, Sara Pieri, Tom Gibbs","submitted_at":"2024-08-29T17:30:05Z","abstract_excerpt":"Large language models (LLMs) are improving at an exceptional rate. However, these models are still susceptible to jailbreak attacks, which are becoming increasingly dangerous as models become increasingly powerful. In this work, we introduce a dataset of jailbreaks where each example can be input in both a single or a multi-turn format. We show that while equivalent in content, they are not equivalent in jailbreak success: defending against one structure does not guarantee defense against the other. Similarly, LLM-based filter guardrails also perform differently depending on not just the input"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.00137","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CR","submitted_at":"2024-08-29T17:30:05Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"a9dca89e72c8c5bac018c0d29a042fae790c301d1d83759720c5c14bc3a05822","abstract_canon_sha256":"ebd63ab79cd48927365794ddc44cbdf6c66ac541ff480db96d5a857194b1c38f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:01:41.473024Z","signature_b64":"wvp84u/rR9EXv7MJTFTwTLLhzAlMjzz6Rtj6hnM2aGAobMZWzl7XlZ0wmWyYBRPM9nxZrdM0dI+nUXms/lejCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"04a704bf168b5fa747416a91ca01af1d7580822390b255acdc8f710cb35d8125","last_reissued_at":"2026-07-05T09:01:41.472566Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:01:41.472566Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Emerging Vulnerabilities in Frontier Models: Multi-Turn Jailbreak Attacks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CR","authors_text":"Ethan Kosak-Hine, George Ingebretsen, Jason Zhang, Julius Broomfield, Kellin Pelrine, Reihaneh Iranmanesh, Reihaneh Rabbany, Sara Pieri, Tom Gibbs","submitted_at":"2024-08-29T17:30:05Z","abstract_excerpt":"Large language models (LLMs) are improving at an exceptional rate. However, these models are still susceptible to jailbreak attacks, which are becoming increasingly dangerous as models become increasingly powerful. In this work, we introduce a dataset of jailbreaks where each example can be input in both a single or a multi-turn format. We show that while equivalent in content, they are not equivalent in jailbreak success: defending against one structure does not guarantee defense against the other. Similarly, LLM-based filter guardrails also perform differently depending on not just the input"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.00137","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.00137/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.00137","created_at":"2026-07-05T09:01:41.472617+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.00137v1","created_at":"2026-07-05T09:01:41.472617+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.00137","created_at":"2026-07-05T09:01:41.472617+00:00"},{"alias_kind":"pith_short_12","alias_value":"ASTQJPYWRNP2","created_at":"2026-07-05T09:01:41.472617+00:00"},{"alias_kind":"pith_short_16","alias_value":"ASTQJPYWRNP2OR2B","created_at":"2026-07-05T09:01:41.472617+00:00"},{"alias_kind":"pith_short_8","alias_value":"ASTQJPYW","created_at":"2026-07-05T09:01:41.472617+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21082","citing_title":"Scalable Hierarchical Attention Transformers for Multi-Turn Jailbreak Detection in Long Conversations","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2410.09024","citing_title":"AgentHarm: A Benchmark for Measuring Harmfulness of LLM Agents","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11309","citing_title":"The Salami Slicing Threat: Exploiting Cumulative Risks in LLM Systems","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ASTQJPYWRNP2OR2BNKI4UANPDV","json":"https://pith.science/pith/ASTQJPYWRNP2OR2BNKI4UANPDV.json","graph_json":"https://pith.science/api/pith-number/ASTQJPYWRNP2OR2BNKI4UANPDV/graph.json","events_json":"https://pith.science/api/pith-number/ASTQJPYWRNP2OR2BNKI4UANPDV/events.json","paper":"https://pith.science/paper/ASTQJPYW"},"agent_actions":{"view_html":"https://pith.science/pith/ASTQJPYWRNP2OR2BNKI4UANPDV","download_json":"https://pith.science/pith/ASTQJPYWRNP2OR2BNKI4UANPDV.json","view_paper":"https://pith.science/paper/ASTQJPYW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.00137&json=true","fetch_graph":"https://pith.science/api/pith-number/ASTQJPYWRNP2OR2BNKI4UANPDV/graph.json","fetch_events":"https://pith.science/api/pith-number/ASTQJPYWRNP2OR2BNKI4UANPDV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ASTQJPYWRNP2OR2BNKI4UANPDV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ASTQJPYWRNP2OR2BNKI4UANPDV/action/storage_attestation","attest_author":"https://pith.science/pith/ASTQJPYWRNP2OR2BNKI4UANPDV/action/author_attestation","sign_citation":"https://pith.science/pith/ASTQJPYWRNP2OR2BNKI4UANPDV/action/citation_signature","submit_replication":"https://pith.science/pith/ASTQJPYWRNP2OR2BNKI4UANPDV/action/replication_record"}},"created_at":"2026-07-05T09:01:41.472617+00:00","updated_at":"2026-07-05T09:01:41.472617+00:00"}