{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:VPOOLXTYF4BDKGCVZ5VM54KGVR","short_pith_number":"pith:VPOOLXTY","schema_version":"1.0","canonical_sha256":"abdce5de782f02351855cf6acef146ac517ea93358688e2407617d1015aebf2b","source":{"kind":"arxiv","id":"2412.17034","version":2},"attestation_state":"computed","paper":{"title":"Shaping the Safety Boundaries: Understanding and Defending Against Jailbreaks in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jiahui Geng, Lang Gao, Preslav Nakov, Xiangliang Zhang, Xiuying Chen","submitted_at":"2024-12-22T14:18:39Z","abstract_excerpt":"Jailbreaking in Large Language Models (LLMs) is a major security concern as it can deceive LLMs to generate harmful text. Yet, there is still insufficient understanding of how jailbreaking works, which makes it hard to develop effective defense strategies. We aim to shed more light into this issue: we conduct a detailed large-scale analysis of seven different jailbreak methods and find that these disagreements stem from insufficient observation samples. In particular, we introduce \\textit{safety boundary}, and we find that jailbreaks shift harmful activations outside that safety boundary, wher"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.17034","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-12-22T14:18:39Z","cross_cats_sorted":[],"title_canon_sha256":"37a5d7f2efbc1dfe555ffaa63b6d6ba2ba94b093234e625e9074459c6316cfd4","abstract_canon_sha256":"5ad15267b278508246a162eb686b5b10f16ef41ca427c11d1bda8a3a1ca9e3a4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:06:54.748820Z","signature_b64":"O7hir/qjfWHNP4dvZi54cKFC3FsHUbMLxGk01hWDinCpuNRVudCehYZVlyeyxyfBhgLqm+CA5cH2zFd17u9JAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"abdce5de782f02351855cf6acef146ac517ea93358688e2407617d1015aebf2b","last_reissued_at":"2026-07-05T11:06:54.748288Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:06:54.748288Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Shaping the Safety Boundaries: Understanding and Defending Against Jailbreaks in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jiahui Geng, Lang Gao, Preslav Nakov, Xiangliang Zhang, Xiuying Chen","submitted_at":"2024-12-22T14:18:39Z","abstract_excerpt":"Jailbreaking in Large Language Models (LLMs) is a major security concern as it can deceive LLMs to generate harmful text. Yet, there is still insufficient understanding of how jailbreaking works, which makes it hard to develop effective defense strategies. We aim to shed more light into this issue: we conduct a detailed large-scale analysis of seven different jailbreak methods and find that these disagreements stem from insufficient observation samples. In particular, we introduce \\textit{safety boundary}, and we find that jailbreaks shift harmful activations outside that safety boundary, wher"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.17034","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.17034/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.17034","created_at":"2026-07-05T11:06:54.748353+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.17034v2","created_at":"2026-07-05T11:06:54.748353+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.17034","created_at":"2026-07-05T11:06:54.748353+00:00"},{"alias_kind":"pith_short_12","alias_value":"VPOOLXTYF4BD","created_at":"2026-07-05T11:06:54.748353+00:00"},{"alias_kind":"pith_short_16","alias_value":"VPOOLXTYF4BDKGCV","created_at":"2026-07-05T11:06:54.748353+00:00"},{"alias_kind":"pith_short_8","alias_value":"VPOOLXTY","created_at":"2026-07-05T11:06:54.748353+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07335","citing_title":"Defending Jailbreak Attacks on Large Language Models via Manifold Trajectory Kinetics","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31748","citing_title":"Addressing Over-Refusal in LLMs with Competing Rewards","ref_index":70,"is_internal_anchor":false},{"citing_arxiv_id":"2510.20129","citing_title":"SAID: Safety-Aware Intent Defense via Prefix Probing for Large Language Models","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VPOOLXTYF4BDKGCVZ5VM54KGVR","json":"https://pith.science/pith/VPOOLXTYF4BDKGCVZ5VM54KGVR.json","graph_json":"https://pith.science/api/pith-number/VPOOLXTYF4BDKGCVZ5VM54KGVR/graph.json","events_json":"https://pith.science/api/pith-number/VPOOLXTYF4BDKGCVZ5VM54KGVR/events.json","paper":"https://pith.science/paper/VPOOLXTY"},"agent_actions":{"view_html":"https://pith.science/pith/VPOOLXTYF4BDKGCVZ5VM54KGVR","download_json":"https://pith.science/pith/VPOOLXTYF4BDKGCVZ5VM54KGVR.json","view_paper":"https://pith.science/paper/VPOOLXTY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.17034&json=true","fetch_graph":"https://pith.science/api/pith-number/VPOOLXTYF4BDKGCVZ5VM54KGVR/graph.json","fetch_events":"https://pith.science/api/pith-number/VPOOLXTYF4BDKGCVZ5VM54KGVR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VPOOLXTYF4BDKGCVZ5VM54KGVR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VPOOLXTYF4BDKGCVZ5VM54KGVR/action/storage_attestation","attest_author":"https://pith.science/pith/VPOOLXTYF4BDKGCVZ5VM54KGVR/action/author_attestation","sign_citation":"https://pith.science/pith/VPOOLXTYF4BDKGCVZ5VM54KGVR/action/citation_signature","submit_replication":"https://pith.science/pith/VPOOLXTYF4BDKGCVZ5VM54KGVR/action/replication_record"}},"created_at":"2026-07-05T11:06:54.748353+00:00","updated_at":"2026-07-05T11:06:54.748353+00:00"}