{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:26ZTRR2F22FANJVBUOTQVLTJMI","short_pith_number":"pith:26ZTRR2F","schema_version":"1.0","canonical_sha256":"d7b338c745d68a06a6a1a3a70aae69622808f9c2c73f6042c6469a6ed110767e","source":{"kind":"arxiv","id":"2404.05880","version":2},"attestation_state":"computed","paper":{"title":"Eraser: Jailbreaking Defense in Large Language Models via Unlearning Harmful Knowledge","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Cen Chen, Huiping Zhuang, Jianwei Wang, Weikai Lu, Zelin Chen, Zhengdong Lu, Ziqian Zeng","submitted_at":"2024-04-08T21:26:22Z","abstract_excerpt":"Jailbreaking attacks can enable Large Language Models (LLMs) to bypass the safeguard and generate harmful content. Existing jailbreaking defense methods have failed to address the fundamental issue that harmful knowledge resides within the model, leading to potential jailbreak risks for LLMs. In this paper, we propose a novel defense method called Eraser, which mainly includes three goals: unlearning harmful knowledge, retaining general knowledge, and maintaining safety alignment. The intuition is that if an LLM forgets the specific knowledge required to answer a harmful question, it will no l"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.05880","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-04-08T21:26:22Z","cross_cats_sorted":[],"title_canon_sha256":"ea0b933138ca0480367ac48901ae492f3871422db40a5c9ac676aeee2cc33106","abstract_canon_sha256":"ea311be10a3a1c437b7236e3a493e6b61f64b51aa558ce9d62b128e35483f1c6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:39:33.340433Z","signature_b64":"iaasHWwIJWEesviTKgm0eBB/6j4+AXCfRHKQ6P8EEsv12HvHkhAg/bddMNIFQDbEIPPfFktnrf6Y5epXVygLBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d7b338c745d68a06a6a1a3a70aae69622808f9c2c73f6042c6469a6ed110767e","last_reissued_at":"2026-07-05T08:39:33.339945Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:39:33.339945Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Eraser: Jailbreaking Defense in Large Language Models via Unlearning Harmful Knowledge","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Cen Chen, Huiping Zhuang, Jianwei Wang, Weikai Lu, Zelin Chen, Zhengdong Lu, Ziqian Zeng","submitted_at":"2024-04-08T21:26:22Z","abstract_excerpt":"Jailbreaking attacks can enable Large Language Models (LLMs) to bypass the safeguard and generate harmful content. Existing jailbreaking defense methods have failed to address the fundamental issue that harmful knowledge resides within the model, leading to potential jailbreak risks for LLMs. In this paper, we propose a novel defense method called Eraser, which mainly includes three goals: unlearning harmful knowledge, retaining general knowledge, and maintaining safety alignment. The intuition is that if an LLM forgets the specific knowledge required to answer a harmful question, it will no l"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.05880","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.05880/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.05880","created_at":"2026-07-05T08:39:33.339993+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.05880v2","created_at":"2026-07-05T08:39:33.339993+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.05880","created_at":"2026-07-05T08:39:33.339993+00:00"},{"alias_kind":"pith_short_12","alias_value":"26ZTRR2F22FA","created_at":"2026-07-05T08:39:33.339993+00:00"},{"alias_kind":"pith_short_16","alias_value":"26ZTRR2F22FANJVB","created_at":"2026-07-05T08:39:33.339993+00:00"},{"alias_kind":"pith_short_8","alias_value":"26ZTRR2F","created_at":"2026-07-05T08:39:33.339993+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.02920","citing_title":"Fast Unlearning at Scale via Margin Self-Correction","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2510.17210","citing_title":"Wisdom is Knowing What not to Say: Hallucination-Free LLMs Unlearning via Attention Shifting","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24082","citing_title":"Jailbreaking Frontier Foundation Models Through Intention Deception","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07727","citing_title":"TrajGuard: Streaming Hidden-state Trajectory Detection for Decoding-time Jailbreak Defense","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06154","citing_title":"Exclusive Unlearning","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/26ZTRR2F22FANJVBUOTQVLTJMI","json":"https://pith.science/pith/26ZTRR2F22FANJVBUOTQVLTJMI.json","graph_json":"https://pith.science/api/pith-number/26ZTRR2F22FANJVBUOTQVLTJMI/graph.json","events_json":"https://pith.science/api/pith-number/26ZTRR2F22FANJVBUOTQVLTJMI/events.json","paper":"https://pith.science/paper/26ZTRR2F"},"agent_actions":{"view_html":"https://pith.science/pith/26ZTRR2F22FANJVBUOTQVLTJMI","download_json":"https://pith.science/pith/26ZTRR2F22FANJVBUOTQVLTJMI.json","view_paper":"https://pith.science/paper/26ZTRR2F","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.05880&json=true","fetch_graph":"https://pith.science/api/pith-number/26ZTRR2F22FANJVBUOTQVLTJMI/graph.json","fetch_events":"https://pith.science/api/pith-number/26ZTRR2F22FANJVBUOTQVLTJMI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/26ZTRR2F22FANJVBUOTQVLTJMI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/26ZTRR2F22FANJVBUOTQVLTJMI/action/storage_attestation","attest_author":"https://pith.science/pith/26ZTRR2F22FANJVBUOTQVLTJMI/action/author_attestation","sign_citation":"https://pith.science/pith/26ZTRR2F22FANJVBUOTQVLTJMI/action/citation_signature","submit_replication":"https://pith.science/pith/26ZTRR2F22FANJVBUOTQVLTJMI/action/replication_record"}},"created_at":"2026-07-05T08:39:33.339993+00:00","updated_at":"2026-07-05T08:39:33.339993+00:00"}