{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:6GFLTDUSCOO3VKPTAKS4D4ZATD","short_pith_number":"pith:6GFLTDUS","schema_version":"1.0","canonical_sha256":"f18ab98e92139dbaa9f302a5c1f32098d5b340b8dfddbb971988dfb76b3ec2d6","source":{"kind":"arxiv","id":"2405.01496","version":1},"attestation_state":"computed","paper":{"title":"LocInv: Localization-aware Inversion for Text-Guided Image Editing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chuanming Tang, Fei Yang, Joost Van De Weijer, Kai Wang","submitted_at":"2024-05-02T17:27:04Z","abstract_excerpt":"Large-scale Text-to-Image (T2I) diffusion models demonstrate significant generation capabilities based on textual prompts. Based on the T2I diffusion models, text-guided image editing research aims to empower users to manipulate generated images by altering the text prompts. However, existing image editing techniques are prone to editing over unintentional regions that are beyond the intended target area, primarily due to inaccuracies in cross-attention maps. To address this problem, we propose Localization-aware Inversion (LocInv), which exploits segmentation maps or bounding boxes as extra l"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.01496","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-05-02T17:27:04Z","cross_cats_sorted":[],"title_canon_sha256":"3e74f4a382aabd50fee2cf2dc7caf05062290c8f03e0a0c0b47dbea95d053353","abstract_canon_sha256":"0b87c9f032f5c85d732bc0cd726dffa3267148f33d46c6b809b3a7443f985b14"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:14:45.895993Z","signature_b64":"oHrZFjUlFq/nJ1uAqSg1sr5ouUEnDHDRc3EGcsts8hST4w/9Co09U3kGKTJ17wgPKafXe3kn2i7ZYaDV1ApCAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f18ab98e92139dbaa9f302a5c1f32098d5b340b8dfddbb971988dfb76b3ec2d6","last_reissued_at":"2026-07-05T08:14:45.895514Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:14:45.895514Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LocInv: Localization-aware Inversion for Text-Guided Image Editing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chuanming Tang, Fei Yang, Joost Van De Weijer, Kai Wang","submitted_at":"2024-05-02T17:27:04Z","abstract_excerpt":"Large-scale Text-to-Image (T2I) diffusion models demonstrate significant generation capabilities based on textual prompts. Based on the T2I diffusion models, text-guided image editing research aims to empower users to manipulate generated images by altering the text prompts. However, existing image editing techniques are prone to editing over unintentional regions that are beyond the intended target area, primarily due to inaccuracies in cross-attention maps. To address this problem, we propose Localization-aware Inversion (LocInv), which exploits segmentation maps or bounding boxes as extra l"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.01496","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.01496/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.01496","created_at":"2026-07-05T08:14:45.895571+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.01496v1","created_at":"2026-07-05T08:14:45.895571+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.01496","created_at":"2026-07-05T08:14:45.895571+00:00"},{"alias_kind":"pith_short_12","alias_value":"6GFLTDUSCOO3","created_at":"2026-07-05T08:14:45.895571+00:00"},{"alias_kind":"pith_short_16","alias_value":"6GFLTDUSCOO3VKPT","created_at":"2026-07-05T08:14:45.895571+00:00"},{"alias_kind":"pith_short_8","alias_value":"6GFLTDUS","created_at":"2026-07-05T08:14:45.895571+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.13558","citing_title":"Edit the Bits, Diff the Codes: Bitwise Residual Editing for Visual Autoregressive Models","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01380","citing_title":"Training-free image inversion for one-step diffusion models","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2506.14399","citing_title":"Factored Classifier-Free Guidance","ref_index":55,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6GFLTDUSCOO3VKPTAKS4D4ZATD","json":"https://pith.science/pith/6GFLTDUSCOO3VKPTAKS4D4ZATD.json","graph_json":"https://pith.science/api/pith-number/6GFLTDUSCOO3VKPTAKS4D4ZATD/graph.json","events_json":"https://pith.science/api/pith-number/6GFLTDUSCOO3VKPTAKS4D4ZATD/events.json","paper":"https://pith.science/paper/6GFLTDUS"},"agent_actions":{"view_html":"https://pith.science/pith/6GFLTDUSCOO3VKPTAKS4D4ZATD","download_json":"https://pith.science/pith/6GFLTDUSCOO3VKPTAKS4D4ZATD.json","view_paper":"https://pith.science/paper/6GFLTDUS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.01496&json=true","fetch_graph":"https://pith.science/api/pith-number/6GFLTDUSCOO3VKPTAKS4D4ZATD/graph.json","fetch_events":"https://pith.science/api/pith-number/6GFLTDUSCOO3VKPTAKS4D4ZATD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6GFLTDUSCOO3VKPTAKS4D4ZATD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6GFLTDUSCOO3VKPTAKS4D4ZATD/action/storage_attestation","attest_author":"https://pith.science/pith/6GFLTDUSCOO3VKPTAKS4D4ZATD/action/author_attestation","sign_citation":"https://pith.science/pith/6GFLTDUSCOO3VKPTAKS4D4ZATD/action/citation_signature","submit_replication":"https://pith.science/pith/6GFLTDUSCOO3VKPTAKS4D4ZATD/action/replication_record"}},"created_at":"2026-07-05T08:14:45.895571+00:00","updated_at":"2026-07-05T08:14:45.895571+00:00"}