{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:UG2KSMONWWNMOI4BY7ZL7CDMQN","short_pith_number":"pith:UG2KSMON","schema_version":"1.0","canonical_sha256":"a1b4a931cdb59ac72381c7f2bf886c83517b50da7cd8f1befdf93265db4ae023","source":{"kind":"arxiv","id":"2403.14472","version":5},"attestation_state":"computed","paper":{"title":"Detoxifying Large Language Models via Knowledge Editing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.HC","cs.LG"],"primary_cat":"cs.CL","authors_text":"Huajun Chen, Jindong Wang, Linyi Yang, Mengru Wang, Ningyu Zhang, Qishen Zhang, Shumin Deng, Yunzhi Yao, Zekun Xi, Ziwen Xu","submitted_at":"2024-03-21T15:18:30Z","abstract_excerpt":"This paper investigates using knowledge editing techniques to detoxify Large Language Models (LLMs). We construct a benchmark, SafeEdit, which covers nine unsafe categories with various powerful attack prompts and equips comprehensive metrics for systematic evaluation. We conduct experiments with several knowledge editing approaches, indicating that knowledge editing has the potential to detoxify LLMs with a limited impact on general performance efficiently. Then, we propose a simple yet effective baseline, dubbed Detoxifying with Intraoperative Neural Monitoring (DINM), to diminish the toxici"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.14472","kind":"arxiv","version":5},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-03-21T15:18:30Z","cross_cats_sorted":["cs.AI","cs.CV","cs.HC","cs.LG"],"title_canon_sha256":"cd36c2eb2f5658d336edc524aaf94e3c6d6a1a37fad6cf4518eff329d91d820c","abstract_canon_sha256":"f02abe628f552f2f30d196d4ce5dff95b1d693ec8a8095455cbf27d32c3ac473"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:24:03.909798Z","signature_b64":"XaU9zfRX7yVHS796MuiZ/1BEN1hiUBXRY4zw/5QQvD9aD5xA9hfrspqTqq0O6sKIFffnpaJZpQcg1qfJtTu4DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a1b4a931cdb59ac72381c7f2bf886c83517b50da7cd8f1befdf93265db4ae023","last_reissued_at":"2026-07-05T08:24:03.909243Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:24:03.909243Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Detoxifying Large Language Models via Knowledge Editing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.HC","cs.LG"],"primary_cat":"cs.CL","authors_text":"Huajun Chen, Jindong Wang, Linyi Yang, Mengru Wang, Ningyu Zhang, Qishen Zhang, Shumin Deng, Yunzhi Yao, Zekun Xi, Ziwen Xu","submitted_at":"2024-03-21T15:18:30Z","abstract_excerpt":"This paper investigates using knowledge editing techniques to detoxify Large Language Models (LLMs). We construct a benchmark, SafeEdit, which covers nine unsafe categories with various powerful attack prompts and equips comprehensive metrics for systematic evaluation. We conduct experiments with several knowledge editing approaches, indicating that knowledge editing has the potential to detoxify LLMs with a limited impact on general performance efficiently. Then, we propose a simple yet effective baseline, dubbed Detoxifying with Intraoperative Neural Monitoring (DINM), to diminish the toxici"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.14472","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.14472/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.14472","created_at":"2026-07-05T08:24:03.909330+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.14472v5","created_at":"2026-07-05T08:24:03.909330+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.14472","created_at":"2026-07-05T08:24:03.909330+00:00"},{"alias_kind":"pith_short_12","alias_value":"UG2KSMONWWNM","created_at":"2026-07-05T08:24:03.909330+00:00"},{"alias_kind":"pith_short_16","alias_value":"UG2KSMONWWNMOI4B","created_at":"2026-07-05T08:24:03.909330+00:00"},{"alias_kind":"pith_short_8","alias_value":"UG2KSMON","created_at":"2026-07-05T08:24:03.909330+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.14087","citing_title":"Measuring and Mitigating Toxicity in Large Language Models: A Comprehensive Replication Study","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2404.13501","citing_title":"A Survey on the Memory Mechanism of Large Language Model based Agents","ref_index":137,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14087","citing_title":"Measuring and Mitigating Toxicity in Large Language Models: A Comprehensive Replication Study","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08846","citing_title":"Dictionary-Aligned Concept Control for Safeguarding Multimodal LLMs","ref_index":104,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UG2KSMONWWNMOI4BY7ZL7CDMQN","json":"https://pith.science/pith/UG2KSMONWWNMOI4BY7ZL7CDMQN.json","graph_json":"https://pith.science/api/pith-number/UG2KSMONWWNMOI4BY7ZL7CDMQN/graph.json","events_json":"https://pith.science/api/pith-number/UG2KSMONWWNMOI4BY7ZL7CDMQN/events.json","paper":"https://pith.science/paper/UG2KSMON"},"agent_actions":{"view_html":"https://pith.science/pith/UG2KSMONWWNMOI4BY7ZL7CDMQN","download_json":"https://pith.science/pith/UG2KSMONWWNMOI4BY7ZL7CDMQN.json","view_paper":"https://pith.science/paper/UG2KSMON","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.14472&json=true","fetch_graph":"https://pith.science/api/pith-number/UG2KSMONWWNMOI4BY7ZL7CDMQN/graph.json","fetch_events":"https://pith.science/api/pith-number/UG2KSMONWWNMOI4BY7ZL7CDMQN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UG2KSMONWWNMOI4BY7ZL7CDMQN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UG2KSMONWWNMOI4BY7ZL7CDMQN/action/storage_attestation","attest_author":"https://pith.science/pith/UG2KSMONWWNMOI4BY7ZL7CDMQN/action/author_attestation","sign_citation":"https://pith.science/pith/UG2KSMONWWNMOI4BY7ZL7CDMQN/action/citation_signature","submit_replication":"https://pith.science/pith/UG2KSMONWWNMOI4BY7ZL7CDMQN/action/replication_record"}},"created_at":"2026-07-05T08:24:03.909330+00:00","updated_at":"2026-07-05T08:24:03.909330+00:00"}