{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:GIYDQMDFIUUKIACCVCFEHDGZED","short_pith_number":"pith:GIYDQMDF","schema_version":"1.0","canonical_sha256":"32303830654528a40042a88a438cd920e8f5d72362b194ddbe95f3b6fc44eaaf","source":{"kind":"arxiv","id":"2504.21307","version":2},"attestation_state":"computed","paper":{"title":"The Dual Power of Interpretable Token Embeddings: Jailbreaking Attacks and Defenses for Diffusion Model Unlearning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Qing Qu, Sijia Liu, Siyi Chen, Yimeng Zhang","submitted_at":"2025-04-30T04:33:43Z","abstract_excerpt":"Despite the remarkable generation capabilities of diffusion models, recent studies have shown that they can memorize and create harmful content when given specific text prompts. Although fine-tuning approaches have been developed to mitigate this issue by unlearning harmful concepts, these methods can be easily circumvented through jailbreaking attacks. This implies that the harmful concept has not been fully erased from the model. However, existing jailbreaking attack methods, while effective, lack interpretability regarding why unlearned models still retain the concept, thereby hindering the"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.21307","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-04-30T04:33:43Z","cross_cats_sorted":[],"title_canon_sha256":"44aca3cc382a601ef76ad8e24d88f19af8e7c357b765e47f8a8220e31efb6f35","abstract_canon_sha256":"a83de39c19fb22ec3f4433ec546c33809b0c837537ad8791872a83e172d6138a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:14:06.674525Z","signature_b64":"ImB9JFkkq3QM8lS7basPSEYEHuKd0WgMcsCNPoFCbIg01BDR0mzgx/pJjz6zinQZzEICyUyutnu5d5mk+sztDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"32303830654528a40042a88a438cd920e8f5d72362b194ddbe95f3b6fc44eaaf","last_reissued_at":"2026-07-05T11:14:06.674056Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:14:06.674056Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Dual Power of Interpretable Token Embeddings: Jailbreaking Attacks and Defenses for Diffusion Model Unlearning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Qing Qu, Sijia Liu, Siyi Chen, Yimeng Zhang","submitted_at":"2025-04-30T04:33:43Z","abstract_excerpt":"Despite the remarkable generation capabilities of diffusion models, recent studies have shown that they can memorize and create harmful content when given specific text prompts. Although fine-tuning approaches have been developed to mitigate this issue by unlearning harmful concepts, these methods can be easily circumvented through jailbreaking attacks. This implies that the harmful concept has not been fully erased from the model. However, existing jailbreaking attack methods, while effective, lack interpretability regarding why unlearned models still retain the concept, thereby hindering the"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.21307","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.21307/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.21307","created_at":"2026-07-05T11:14:06.674112+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.21307v2","created_at":"2026-07-05T11:14:06.674112+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.21307","created_at":"2026-07-05T11:14:06.674112+00:00"},{"alias_kind":"pith_short_12","alias_value":"GIYDQMDFIUUK","created_at":"2026-07-05T11:14:06.674112+00:00"},{"alias_kind":"pith_short_16","alias_value":"GIYDQMDFIUUKIACC","created_at":"2026-07-05T11:14:06.674112+00:00"},{"alias_kind":"pith_short_8","alias_value":"GIYDQMDF","created_at":"2026-07-05T11:14:06.674112+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07907","citing_title":"Multimodal Unlearning Across Vision, Language, Video, and Audio: Survey of Methods, Datasets, and Benchmarks","ref_index":156,"is_internal_anchor":true},{"citing_arxiv_id":"2606.06624","citing_title":"Principles and Practice of Deep Representation Learning: or a Mathematical Theory of Memory","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10600","citing_title":"Generate \"Normal\", Edit Poisoned: Branding Injection via Hint Embedding in Image Editing","ref_index":31,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GIYDQMDFIUUKIACCVCFEHDGZED","json":"https://pith.science/pith/GIYDQMDFIUUKIACCVCFEHDGZED.json","graph_json":"https://pith.science/api/pith-number/GIYDQMDFIUUKIACCVCFEHDGZED/graph.json","events_json":"https://pith.science/api/pith-number/GIYDQMDFIUUKIACCVCFEHDGZED/events.json","paper":"https://pith.science/paper/GIYDQMDF"},"agent_actions":{"view_html":"https://pith.science/pith/GIYDQMDFIUUKIACCVCFEHDGZED","download_json":"https://pith.science/pith/GIYDQMDFIUUKIACCVCFEHDGZED.json","view_paper":"https://pith.science/paper/GIYDQMDF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.21307&json=true","fetch_graph":"https://pith.science/api/pith-number/GIYDQMDFIUUKIACCVCFEHDGZED/graph.json","fetch_events":"https://pith.science/api/pith-number/GIYDQMDFIUUKIACCVCFEHDGZED/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GIYDQMDFIUUKIACCVCFEHDGZED/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GIYDQMDFIUUKIACCVCFEHDGZED/action/storage_attestation","attest_author":"https://pith.science/pith/GIYDQMDFIUUKIACCVCFEHDGZED/action/author_attestation","sign_citation":"https://pith.science/pith/GIYDQMDFIUUKIACCVCFEHDGZED/action/citation_signature","submit_replication":"https://pith.science/pith/GIYDQMDFIUUKIACCVCFEHDGZED/action/replication_record"}},"created_at":"2026-07-05T11:14:06.674112+00:00","updated_at":"2026-07-05T11:14:06.674112+00:00"}