{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:6PSHH33WABPSAEQJRGYWY7B7WV","short_pith_number":"pith:6PSHH33W","schema_version":"1.0","canonical_sha256":"f3e473ef76005f20120989b16c7c3fb57b2beaf9ba60d9820429565d6f04624d","source":{"kind":"arxiv","id":"2402.13457","version":2},"attestation_state":"computed","paper":{"title":"A Comprehensive Study of Jailbreak Attack versus Defense for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CR","authors_text":"Gelei Deng, Stjepan Picek, Yi Liu, Yuekang Li, Zihao Xu","submitted_at":"2024-02-21T01:26:39Z","abstract_excerpt":"Large Language Models (LLMS) have increasingly become central to generating content with potential societal impacts. Notably, these models have demonstrated capabilities for generating content that could be deemed harmful. To mitigate these risks, researchers have adopted safety training techniques to align model outputs with societal values to curb the generation of malicious content. However, the phenomenon of \"jailbreaking\", where carefully crafted prompts elicit harmful responses from models, persists as a significant challenge. This research conducts a comprehensive analysis of existing s"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.13457","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CR","submitted_at":"2024-02-21T01:26:39Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"5a3533cf19ba349f5a5d1d89356cad8a9819565f94e14075312c7c64e3efb016","abstract_canon_sha256":"d33b490d797d5cbf0d00efc090067a9974094b864b228f6d0e8321c02095c4c5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:20:10.352138Z","signature_b64":"TTr79+rcK2A5gP6MS2nesmwJppHeuGj4lDLhVBnU0dAxlzzq8ubwYjafpgLJPpkrBBCJnrU5sWtoGFdHlx2XAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f3e473ef76005f20120989b16c7c3fb57b2beaf9ba60d9820429565d6f04624d","last_reissued_at":"2026-07-05T08:20:10.351625Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:20:10.351625Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Comprehensive Study of Jailbreak Attack versus Defense for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CR","authors_text":"Gelei Deng, Stjepan Picek, Yi Liu, Yuekang Li, Zihao Xu","submitted_at":"2024-02-21T01:26:39Z","abstract_excerpt":"Large Language Models (LLMS) have increasingly become central to generating content with potential societal impacts. Notably, these models have demonstrated capabilities for generating content that could be deemed harmful. To mitigate these risks, researchers have adopted safety training techniques to align model outputs with societal values to curb the generation of malicious content. However, the phenomenon of \"jailbreaking\", where carefully crafted prompts elicit harmful responses from models, persists as a significant challenge. This research conducts a comprehensive analysis of existing s"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.13457","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.13457/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.13457","created_at":"2026-07-05T08:20:10.351682+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.13457v2","created_at":"2026-07-05T08:20:10.351682+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.13457","created_at":"2026-07-05T08:20:10.351682+00:00"},{"alias_kind":"pith_short_12","alias_value":"6PSHH33WABPS","created_at":"2026-07-05T08:20:10.351682+00:00"},{"alias_kind":"pith_short_16","alias_value":"6PSHH33WABPSAEQJ","created_at":"2026-07-05T08:20:10.351682+00:00"},{"alias_kind":"pith_short_8","alias_value":"6PSHH33W","created_at":"2026-07-05T08:20:10.351682+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.00150","citing_title":"Persona Attack: Incremental Memory Injection Jailbreak Attack against Large Language Models","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2508.11222","citing_title":"ORFuzz: Fuzzing the \"Other Side\" of LLM Safety -- Testing Over-Refusal","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2406.11717","citing_title":"Refusal in Language Models Is Mediated by a Single Direction","ref_index":201,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11309","citing_title":"The Salami Slicing Threat: Exploiting Cumulative Risks in LLM Systems","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09747","citing_title":"ADAM: A Systematic Data Extraction Attack on Agent Memory via Adaptive Querying","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6PSHH33WABPSAEQJRGYWY7B7WV","json":"https://pith.science/pith/6PSHH33WABPSAEQJRGYWY7B7WV.json","graph_json":"https://pith.science/api/pith-number/6PSHH33WABPSAEQJRGYWY7B7WV/graph.json","events_json":"https://pith.science/api/pith-number/6PSHH33WABPSAEQJRGYWY7B7WV/events.json","paper":"https://pith.science/paper/6PSHH33W"},"agent_actions":{"view_html":"https://pith.science/pith/6PSHH33WABPSAEQJRGYWY7B7WV","download_json":"https://pith.science/pith/6PSHH33WABPSAEQJRGYWY7B7WV.json","view_paper":"https://pith.science/paper/6PSHH33W","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.13457&json=true","fetch_graph":"https://pith.science/api/pith-number/6PSHH33WABPSAEQJRGYWY7B7WV/graph.json","fetch_events":"https://pith.science/api/pith-number/6PSHH33WABPSAEQJRGYWY7B7WV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6PSHH33WABPSAEQJRGYWY7B7WV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6PSHH33WABPSAEQJRGYWY7B7WV/action/storage_attestation","attest_author":"https://pith.science/pith/6PSHH33WABPSAEQJRGYWY7B7WV/action/author_attestation","sign_citation":"https://pith.science/pith/6PSHH33WABPSAEQJRGYWY7B7WV/action/citation_signature","submit_replication":"https://pith.science/pith/6PSHH33WABPSAEQJRGYWY7B7WV/action/replication_record"}},"created_at":"2026-07-05T08:20:10.351682+00:00","updated_at":"2026-07-05T08:20:10.351682+00:00"}