{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:EXR2Q7AAZBB32W6FFM2X226NUQ","short_pith_number":"pith:EXR2Q7AA","schema_version":"1.0","canonical_sha256":"25e3a87c00c843bd5bc52b357d6bcda43e615a2911b549a0ef233c7d66f665e4","source":{"kind":"arxiv","id":"2406.05498","version":3},"attestation_state":"computed","paper":{"title":"SelfDefend: LLMs Can Defend Themselves against Jailbreaking in a Practical Manner","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CR","authors_text":"Daoyuan Wu, Juergen Rahmel, Ning Liu, Pingchuan Ma, Shuai Wang, Xunguang Wang, Yang Liu, Yingjiu Li, Zhenlan Ji, Zongjie Li","submitted_at":"2024-06-08T15:45:31Z","abstract_excerpt":"Jailbreaking is an emerging adversarial attack that bypasses the safety alignment deployed in off-the-shelf large language models (LLMs) and has evolved into multiple categories: human-based, optimization-based, generation-based, and the recent indirect and multilingual jailbreaks. However, delivering a practical jailbreak defense is challenging because it needs to not only handle all the above jailbreak attacks but also incur negligible delays to user prompts, as well as be compatible with both open-source and closed-source LLMs. Inspired by how the traditional security concept of shadow stac"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.05498","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CR","submitted_at":"2024-06-08T15:45:31Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"28cc0b37cdddb319cdae61fb08e125a04735355ac1915a6382a80a568c575f83","abstract_canon_sha256":"4a46d726b1a371c17837387eed71c1f8e388f4e04a09bacc556088bdae301d8b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:09:44.501265Z","signature_b64":"/jvV7hRoFGyocm+qBnDYCIxnjTtMhUvHjP0YK5M3dRFeh3i7eAHWxIUJBcaLaVKKu9eyxIhXZo6SXBV119dGAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"25e3a87c00c843bd5bc52b357d6bcda43e615a2911b549a0ef233c7d66f665e4","last_reissued_at":"2026-07-05T10:09:44.500798Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:09:44.500798Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SelfDefend: LLMs Can Defend Themselves against Jailbreaking in a Practical Manner","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CR","authors_text":"Daoyuan Wu, Juergen Rahmel, Ning Liu, Pingchuan Ma, Shuai Wang, Xunguang Wang, Yang Liu, Yingjiu Li, Zhenlan Ji, Zongjie Li","submitted_at":"2024-06-08T15:45:31Z","abstract_excerpt":"Jailbreaking is an emerging adversarial attack that bypasses the safety alignment deployed in off-the-shelf large language models (LLMs) and has evolved into multiple categories: human-based, optimization-based, generation-based, and the recent indirect and multilingual jailbreaks. However, delivering a practical jailbreak defense is challenging because it needs to not only handle all the above jailbreak attacks but also incur negligible delays to user prompts, as well as be compatible with both open-source and closed-source LLMs. Inspired by how the traditional security concept of shadow stac"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.05498","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.05498/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.05498","created_at":"2026-07-05T10:09:44.500869+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.05498v3","created_at":"2026-07-05T10:09:44.500869+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.05498","created_at":"2026-07-05T10:09:44.500869+00:00"},{"alias_kind":"pith_short_12","alias_value":"EXR2Q7AAZBB3","created_at":"2026-07-05T10:09:44.500869+00:00"},{"alias_kind":"pith_short_16","alias_value":"EXR2Q7AAZBB32W6F","created_at":"2026-07-05T10:09:44.500869+00:00"},{"alias_kind":"pith_short_8","alias_value":"EXR2Q7AA","created_at":"2026-07-05T10:09:44.500869+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01277","citing_title":"Cognitive Firewall: A Proactive, Zero-Trust, Multi-Gate Framework for LLM Safety","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2502.05206","citing_title":"Safety at Scale: A Comprehensive Survey of Large Model and Agent Safety","ref_index":103,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18248","citing_title":"Beyond Pattern Matching: Seven Cross-Domain Techniques for Prompt Injection Detection","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09225","citing_title":"The Art of the Jailbreak: Formulating Jailbreak Attacks for LLM Security Beyond Binary Scoring","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18248","citing_title":"Beyond Pattern Matching: Seven Cross-Domain Techniques for Prompt Injection Detection","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EXR2Q7AAZBB32W6FFM2X226NUQ","json":"https://pith.science/pith/EXR2Q7AAZBB32W6FFM2X226NUQ.json","graph_json":"https://pith.science/api/pith-number/EXR2Q7AAZBB32W6FFM2X226NUQ/graph.json","events_json":"https://pith.science/api/pith-number/EXR2Q7AAZBB32W6FFM2X226NUQ/events.json","paper":"https://pith.science/paper/EXR2Q7AA"},"agent_actions":{"view_html":"https://pith.science/pith/EXR2Q7AAZBB32W6FFM2X226NUQ","download_json":"https://pith.science/pith/EXR2Q7AAZBB32W6FFM2X226NUQ.json","view_paper":"https://pith.science/paper/EXR2Q7AA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.05498&json=true","fetch_graph":"https://pith.science/api/pith-number/EXR2Q7AAZBB32W6FFM2X226NUQ/graph.json","fetch_events":"https://pith.science/api/pith-number/EXR2Q7AAZBB32W6FFM2X226NUQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EXR2Q7AAZBB32W6FFM2X226NUQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EXR2Q7AAZBB32W6FFM2X226NUQ/action/storage_attestation","attest_author":"https://pith.science/pith/EXR2Q7AAZBB32W6FFM2X226NUQ/action/author_attestation","sign_citation":"https://pith.science/pith/EXR2Q7AAZBB32W6FFM2X226NUQ/action/citation_signature","submit_replication":"https://pith.science/pith/EXR2Q7AAZBB32W6FFM2X226NUQ/action/replication_record"}},"created_at":"2026-07-05T10:09:44.500869+00:00","updated_at":"2026-07-05T10:09:44.500869+00:00"}