{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:YFFWS2GKEXM2GW5RNZUAW5DY2X","short_pith_number":"pith:YFFWS2GK","schema_version":"1.0","canonical_sha256":"c14b6968ca25d9a35bb16e680b7478d5f43a9777d7da40a8ecf97ab57a21e78e","source":{"kind":"arxiv","id":"2502.02958","version":3},"attestation_state":"computed","paper":{"title":"Position: Editing Large Language Models Poses Serious Safety Risks","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Christin Seifert, Daniel Braun, J\\\"org Schl\\\"otterer, Paul Youssef, Zhixue Zhao","submitted_at":"2025-02-05T07:51:32Z","abstract_excerpt":"Large Language Models (LLMs) contain large amounts of facts about the world. These facts can become outdated over time, which has led to the development of knowledge editing methods (KEs) that can change specific facts in LLMs with limited side effects. This position paper argues that editing LLMs poses serious safety risks that have been largely overlooked. First, we note the fact that KEs are widely available, computationally inexpensive, highly performant, and stealthy makes them an attractive tool for malicious actors. Second, we discuss malicious use cases of KEs, showing how KEs can be e"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.02958","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2025-02-05T07:51:32Z","cross_cats_sorted":[],"title_canon_sha256":"dd2e477dc78af7fc06657a9f249290cadea1550466f8f75b06ebf8529cae1808","abstract_canon_sha256":"1216da26902df806a53ef4b02dbc47e38cb2a523c0bfa81649743d041a1b1195"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:22:36.307557Z","signature_b64":"RYcudL04bZ23lAr56K2gADF/lv/k3/jBwTka+HvzaHi29MlKYeZPYjR44undlgeSY3wBZAEUakG4qthQdB0kCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c14b6968ca25d9a35bb16e680b7478d5f43a9777d7da40a8ecf97ab57a21e78e","last_reissued_at":"2026-07-05T11:22:36.306952Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:22:36.306952Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Position: Editing Large Language Models Poses Serious Safety Risks","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Christin Seifert, Daniel Braun, J\\\"org Schl\\\"otterer, Paul Youssef, Zhixue Zhao","submitted_at":"2025-02-05T07:51:32Z","abstract_excerpt":"Large Language Models (LLMs) contain large amounts of facts about the world. These facts can become outdated over time, which has led to the development of knowledge editing methods (KEs) that can change specific facts in LLMs with limited side effects. This position paper argues that editing LLMs poses serious safety risks that have been largely overlooked. First, we note the fact that KEs are widely available, computationally inexpensive, highly performant, and stealthy makes them an attractive tool for malicious actors. Second, we discuss malicious use cases of KEs, showing how KEs can be e"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.02958","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.02958/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.02958","created_at":"2026-07-05T11:22:36.307022+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.02958v3","created_at":"2026-07-05T11:22:36.307022+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.02958","created_at":"2026-07-05T11:22:36.307022+00:00"},{"alias_kind":"pith_short_12","alias_value":"YFFWS2GKEXM2","created_at":"2026-07-05T11:22:36.307022+00:00"},{"alias_kind":"pith_short_16","alias_value":"YFFWS2GKEXM2GW5R","created_at":"2026-07-05T11:22:36.307022+00:00"},{"alias_kind":"pith_short_8","alias_value":"YFFWS2GK","created_at":"2026-07-05T11:22:36.307022+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.01266","citing_title":"Detoxification of Large Language Models through Output-layer Fusion with a Calibration Model","ref_index":47,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YFFWS2GKEXM2GW5RNZUAW5DY2X","json":"https://pith.science/pith/YFFWS2GKEXM2GW5RNZUAW5DY2X.json","graph_json":"https://pith.science/api/pith-number/YFFWS2GKEXM2GW5RNZUAW5DY2X/graph.json","events_json":"https://pith.science/api/pith-number/YFFWS2GKEXM2GW5RNZUAW5DY2X/events.json","paper":"https://pith.science/paper/YFFWS2GK"},"agent_actions":{"view_html":"https://pith.science/pith/YFFWS2GKEXM2GW5RNZUAW5DY2X","download_json":"https://pith.science/pith/YFFWS2GKEXM2GW5RNZUAW5DY2X.json","view_paper":"https://pith.science/paper/YFFWS2GK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.02958&json=true","fetch_graph":"https://pith.science/api/pith-number/YFFWS2GKEXM2GW5RNZUAW5DY2X/graph.json","fetch_events":"https://pith.science/api/pith-number/YFFWS2GKEXM2GW5RNZUAW5DY2X/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YFFWS2GKEXM2GW5RNZUAW5DY2X/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YFFWS2GKEXM2GW5RNZUAW5DY2X/action/storage_attestation","attest_author":"https://pith.science/pith/YFFWS2GKEXM2GW5RNZUAW5DY2X/action/author_attestation","sign_citation":"https://pith.science/pith/YFFWS2GKEXM2GW5RNZUAW5DY2X/action/citation_signature","submit_replication":"https://pith.science/pith/YFFWS2GKEXM2GW5RNZUAW5DY2X/action/replication_record"}},"created_at":"2026-07-05T11:22:36.307022+00:00","updated_at":"2026-07-05T11:22:36.307022+00:00"}