{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:3TUJQ6VGNGV7RY7UK5GKKRTSS7","short_pith_number":"pith:3TUJQ6VG","schema_version":"1.0","canonical_sha256":"dce8987aa669abf8e3f4574ca5467297cb41798f170bf12d0c952039a53c0bf4","source":{"kind":"arxiv","id":"2307.02421","version":2},"attestation_state":"computed","paper":{"title":"DragonDiffusion: Enabling Drag-style Manipulation on Diffusion Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chong Mou, Jian Zhang, Jiechong Song, Xintao Wang, Ying Shan","submitted_at":"2023-07-05T16:43:56Z","abstract_excerpt":"Despite the ability of existing large-scale text-to-image (T2I) models to generate high-quality images from detailed textual descriptions, they often lack the ability to precisely edit the generated or real images. In this paper, we propose a novel image editing method, DragonDiffusion, enabling Drag-style manipulation on Diffusion models. Specifically, we construct classifier guidance based on the strong correspondence of intermediate features in the diffusion model. It can transform the editing signals into gradients via feature correspondence loss to modify the intermediate representation o"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.02421","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-07-05T16:43:56Z","cross_cats_sorted":[],"title_canon_sha256":"f75b93c4eeda26b9e14dcf93f3c6a9f7754b0a27d34762dfde8c5de9a30a07c2","abstract_canon_sha256":"2a315daf2f0fea173dc7c217495e7a80d2fa26236eb99d79e2a65a08b20d8e96"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:14:14.890189Z","signature_b64":"jLYuodGkvi1CQjGoFlgnOR4QNmdXUPkaPfxi7A7VnrkhzatKRDebcMUJpgyTtS3KlXY3rm07hCi8ZYjePLdhCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dce8987aa669abf8e3f4574ca5467297cb41798f170bf12d0c952039a53c0bf4","last_reissued_at":"2026-07-05T07:14:14.889720Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:14:14.889720Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DragonDiffusion: Enabling Drag-style Manipulation on Diffusion Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chong Mou, Jian Zhang, Jiechong Song, Xintao Wang, Ying Shan","submitted_at":"2023-07-05T16:43:56Z","abstract_excerpt":"Despite the ability of existing large-scale text-to-image (T2I) models to generate high-quality images from detailed textual descriptions, they often lack the ability to precisely edit the generated or real images. In this paper, we propose a novel image editing method, DragonDiffusion, enabling Drag-style manipulation on Diffusion models. Specifically, we construct classifier guidance based on the strong correspondence of intermediate features in the diffusion model. It can transform the editing signals into gradients via feature correspondence loss to modify the intermediate representation o"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.02421","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.02421/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.02421","created_at":"2026-07-05T07:14:14.889781+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.02421v2","created_at":"2026-07-05T07:14:14.889781+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.02421","created_at":"2026-07-05T07:14:14.889781+00:00"},{"alias_kind":"pith_short_12","alias_value":"3TUJQ6VGNGV7","created_at":"2026-07-05T07:14:14.889781+00:00"},{"alias_kind":"pith_short_16","alias_value":"3TUJQ6VGNGV7RY7U","created_at":"2026-07-05T07:14:14.889781+00:00"},{"alias_kind":"pith_short_8","alias_value":"3TUJQ6VG","created_at":"2026-07-05T07:14:14.889781+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.24509","citing_title":"{\\Phi}-Noise: Training-Free Temporal Video Conditioning via Phase-Based Noise Manipulation","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2411.14295","citing_title":"DissolveStereo: Coarse Depth Injection for Zero-Shot Stereo Video Generation","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20777","citing_title":"AttriStory: Fine-grained Attribute Realization for Visual Storytelling with Diffusion Models","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15181","citing_title":"From Plans to Pixels: Learning to Plan and Orchestrate for Open-Ended Image Editing","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12575","citing_title":"StructDiff: A Structure-Preserving and Spatially Controllable Diffusion Model for Single-Image Generation","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3TUJQ6VGNGV7RY7UK5GKKRTSS7","json":"https://pith.science/pith/3TUJQ6VGNGV7RY7UK5GKKRTSS7.json","graph_json":"https://pith.science/api/pith-number/3TUJQ6VGNGV7RY7UK5GKKRTSS7/graph.json","events_json":"https://pith.science/api/pith-number/3TUJQ6VGNGV7RY7UK5GKKRTSS7/events.json","paper":"https://pith.science/paper/3TUJQ6VG"},"agent_actions":{"view_html":"https://pith.science/pith/3TUJQ6VGNGV7RY7UK5GKKRTSS7","download_json":"https://pith.science/pith/3TUJQ6VGNGV7RY7UK5GKKRTSS7.json","view_paper":"https://pith.science/paper/3TUJQ6VG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.02421&json=true","fetch_graph":"https://pith.science/api/pith-number/3TUJQ6VGNGV7RY7UK5GKKRTSS7/graph.json","fetch_events":"https://pith.science/api/pith-number/3TUJQ6VGNGV7RY7UK5GKKRTSS7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3TUJQ6VGNGV7RY7UK5GKKRTSS7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3TUJQ6VGNGV7RY7UK5GKKRTSS7/action/storage_attestation","attest_author":"https://pith.science/pith/3TUJQ6VGNGV7RY7UK5GKKRTSS7/action/author_attestation","sign_citation":"https://pith.science/pith/3TUJQ6VGNGV7RY7UK5GKKRTSS7/action/citation_signature","submit_replication":"https://pith.science/pith/3TUJQ6VGNGV7RY7UK5GKKRTSS7/action/replication_record"}},"created_at":"2026-07-05T07:14:14.889781+00:00","updated_at":"2026-07-05T07:14:14.889781+00:00"}