{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:YP2IN6J3ILPIKUWMSBYXBE4MU7","short_pith_number":"pith:YP2IN6J3","schema_version":"1.0","canonical_sha256":"c3f486f93b42de8552cc907170938ca7f71ebcb5f090bddb53343b7828cb245c","source":{"kind":"arxiv","id":"2412.08201","version":1},"attestation_state":"computed","paper":{"title":"Model-Editing-Based Jailbreak against Safety-aligned Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CR","authors_text":"Haoyu Wang, Kailong Wang, Ling Shi, Yuxi Li, Zhibo Zhang","submitted_at":"2024-12-11T08:44:15Z","abstract_excerpt":"Large Language Models (LLMs) have transformed numerous fields by enabling advanced natural language interactions but remain susceptible to critical vulnerabilities, particularly jailbreak attacks. Current jailbreak techniques, while effective, often depend on input modifications, making them detectable and limiting their stealth and scalability. This paper presents Targeted Model Editing (TME), a novel white-box approach that bypasses safety filters by minimally altering internal model structures while preserving the model's intended functionalities. TME identifies and removes safety-critical "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.08201","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CR","submitted_at":"2024-12-11T08:44:15Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"521f4fa8ee4ffc95afe947cfe7675ff965b6c3ca5329a81d5426440014efd790","abstract_canon_sha256":"e7dbcf765d112bb17d8c68a820a80676017cc69eb7e18510c0f73a7b2d3b826a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:47:45.931751Z","signature_b64":"XIuKzd+EpfMWXf2ugwFR7DFuhmbfuzqloJyn9IUNxdqNhfvn35SWHSBiDEV506KkuxGm5fT07FNvACaFUJh3CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c3f486f93b42de8552cc907170938ca7f71ebcb5f090bddb53343b7828cb245c","last_reissued_at":"2026-07-05T09:47:45.931353Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:47:45.931353Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Model-Editing-Based Jailbreak against Safety-aligned Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CR","authors_text":"Haoyu Wang, Kailong Wang, Ling Shi, Yuxi Li, Zhibo Zhang","submitted_at":"2024-12-11T08:44:15Z","abstract_excerpt":"Large Language Models (LLMs) have transformed numerous fields by enabling advanced natural language interactions but remain susceptible to critical vulnerabilities, particularly jailbreak attacks. Current jailbreak techniques, while effective, often depend on input modifications, making them detectable and limiting their stealth and scalability. This paper presents Targeted Model Editing (TME), a novel white-box approach that bypasses safety filters by minimally altering internal model structures while preserving the model's intended functionalities. TME identifies and removes safety-critical "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.08201","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.08201/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.08201","created_at":"2026-07-05T09:47:45.931403+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.08201v1","created_at":"2026-07-05T09:47:45.931403+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.08201","created_at":"2026-07-05T09:47:45.931403+00:00"},{"alias_kind":"pith_short_12","alias_value":"YP2IN6J3ILPI","created_at":"2026-07-05T09:47:45.931403+00:00"},{"alias_kind":"pith_short_16","alias_value":"YP2IN6J3ILPIKUWM","created_at":"2026-07-05T09:47:45.931403+00:00"},{"alias_kind":"pith_short_8","alias_value":"YP2IN6J3","created_at":"2026-07-05T09:47:45.931403+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.05775","citing_title":"Guardians and Offenders: A Survey on Harmful Content Generation and Safety Mitigation of LLM","ref_index":232,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YP2IN6J3ILPIKUWMSBYXBE4MU7","json":"https://pith.science/pith/YP2IN6J3ILPIKUWMSBYXBE4MU7.json","graph_json":"https://pith.science/api/pith-number/YP2IN6J3ILPIKUWMSBYXBE4MU7/graph.json","events_json":"https://pith.science/api/pith-number/YP2IN6J3ILPIKUWMSBYXBE4MU7/events.json","paper":"https://pith.science/paper/YP2IN6J3"},"agent_actions":{"view_html":"https://pith.science/pith/YP2IN6J3ILPIKUWMSBYXBE4MU7","download_json":"https://pith.science/pith/YP2IN6J3ILPIKUWMSBYXBE4MU7.json","view_paper":"https://pith.science/paper/YP2IN6J3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.08201&json=true","fetch_graph":"https://pith.science/api/pith-number/YP2IN6J3ILPIKUWMSBYXBE4MU7/graph.json","fetch_events":"https://pith.science/api/pith-number/YP2IN6J3ILPIKUWMSBYXBE4MU7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YP2IN6J3ILPIKUWMSBYXBE4MU7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YP2IN6J3ILPIKUWMSBYXBE4MU7/action/storage_attestation","attest_author":"https://pith.science/pith/YP2IN6J3ILPIKUWMSBYXBE4MU7/action/author_attestation","sign_citation":"https://pith.science/pith/YP2IN6J3ILPIKUWMSBYXBE4MU7/action/citation_signature","submit_replication":"https://pith.science/pith/YP2IN6J3ILPIKUWMSBYXBE4MU7/action/replication_record"}},"created_at":"2026-07-05T09:47:45.931403+00:00","updated_at":"2026-07-05T09:47:45.931403+00:00"}