{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:KT3NJ72GOGQOOGF6GVIJAR6ZVK","short_pith_number":"pith:KT3NJ72G","schema_version":"1.0","canonical_sha256":"54f6d4ff4671a0e718be35509047d9aaa1e3717e92ff53b2f00cf25558b0300a","source":{"kind":"arxiv","id":"2305.18047","version":1},"attestation_state":"computed","paper":{"title":"InstructEdit: Improving Automatic Masks for Diffusion-based Image Editing With User Instructions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Biao Zhang, Michael Birsak, Peter Wonka, Qian Wang","submitted_at":"2023-05-29T12:24:58Z","abstract_excerpt":"Recent works have explored text-guided image editing using diffusion models and generated edited images based on text prompts. However, the models struggle to accurately locate the regions to be edited and faithfully perform precise edits. In this work, we propose a framework termed InstructEdit that can do fine-grained editing based on user instructions. Our proposed framework has three components: language processor, segmenter, and image editor. The first component, the language processor, processes the user instruction using a large language model. The goal of this processing is to parse th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.18047","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-05-29T12:24:58Z","cross_cats_sorted":[],"title_canon_sha256":"d84232389af6cd17e0428efc71bde7844eeef139573440cd234e8d733142d1e5","abstract_canon_sha256":"21dcfe1948ff940b05e9ec4b62359e51cb21c7f2971aaf3aa21fb5d2b4ad0f5d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:14:51.808698Z","signature_b64":"dYXzHW2wddpr+S7W095kl6vsLO3xuWGGDmy8R7cUo5oScKlWIfQMM8dwHqs3mpeKJHaOEpqMswskt8oNKNmGCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"54f6d4ff4671a0e718be35509047d9aaa1e3717e92ff53b2f00cf25558b0300a","last_reissued_at":"2026-07-05T06:14:51.808245Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:14:51.808245Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"InstructEdit: Improving Automatic Masks for Diffusion-based Image Editing With User Instructions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Biao Zhang, Michael Birsak, Peter Wonka, Qian Wang","submitted_at":"2023-05-29T12:24:58Z","abstract_excerpt":"Recent works have explored text-guided image editing using diffusion models and generated edited images based on text prompts. However, the models struggle to accurately locate the regions to be edited and faithfully perform precise edits. In this work, we propose a framework termed InstructEdit that can do fine-grained editing based on user instructions. Our proposed framework has three components: language processor, segmenter, and image editor. The first component, the language processor, processes the user instruction using a large language model. The goal of this processing is to parse th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.18047","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.18047/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.18047","created_at":"2026-07-05T06:14:51.808297+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.18047v1","created_at":"2026-07-05T06:14:51.808297+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.18047","created_at":"2026-07-05T06:14:51.808297+00:00"},{"alias_kind":"pith_short_12","alias_value":"KT3NJ72GOGQO","created_at":"2026-07-05T06:14:51.808297+00:00"},{"alias_kind":"pith_short_16","alias_value":"KT3NJ72GOGQOOGF6","created_at":"2026-07-05T06:14:51.808297+00:00"},{"alias_kind":"pith_short_8","alias_value":"KT3NJ72G","created_at":"2026-07-05T06:14:51.808297+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.16951","citing_title":"Edit-GRPO: A Locality-Preserving Policy Optimization Framework for Image Editing","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2506.21546","citing_title":"Counterfactual Segmentation Reasoning: Diagnosing and Mitigating Pixel-Grounding Hallucination","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2504.17761","citing_title":"Step1X-Edit: A Practical Framework for General Image Editing","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05180","citing_title":"MIRAGE: Benchmarking and Aligning Multi-Instance Image Editing","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20258","citing_title":"Rethinking Where to Edit: Task-Aware Localization for Instruction-Based Image Editing","ref_index":34,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KT3NJ72GOGQOOGF6GVIJAR6ZVK","json":"https://pith.science/pith/KT3NJ72GOGQOOGF6GVIJAR6ZVK.json","graph_json":"https://pith.science/api/pith-number/KT3NJ72GOGQOOGF6GVIJAR6ZVK/graph.json","events_json":"https://pith.science/api/pith-number/KT3NJ72GOGQOOGF6GVIJAR6ZVK/events.json","paper":"https://pith.science/paper/KT3NJ72G"},"agent_actions":{"view_html":"https://pith.science/pith/KT3NJ72GOGQOOGF6GVIJAR6ZVK","download_json":"https://pith.science/pith/KT3NJ72GOGQOOGF6GVIJAR6ZVK.json","view_paper":"https://pith.science/paper/KT3NJ72G","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.18047&json=true","fetch_graph":"https://pith.science/api/pith-number/KT3NJ72GOGQOOGF6GVIJAR6ZVK/graph.json","fetch_events":"https://pith.science/api/pith-number/KT3NJ72GOGQOOGF6GVIJAR6ZVK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KT3NJ72GOGQOOGF6GVIJAR6ZVK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KT3NJ72GOGQOOGF6GVIJAR6ZVK/action/storage_attestation","attest_author":"https://pith.science/pith/KT3NJ72GOGQOOGF6GVIJAR6ZVK/action/author_attestation","sign_citation":"https://pith.science/pith/KT3NJ72GOGQOOGF6GVIJAR6ZVK/action/citation_signature","submit_replication":"https://pith.science/pith/KT3NJ72GOGQOOGF6GVIJAR6ZVK/action/replication_record"}},"created_at":"2026-07-05T06:14:51.808297+00:00","updated_at":"2026-07-05T06:14:51.808297+00:00"}