{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:Q34QCN3K5SMQN4KBDPQYDO3J44","short_pith_number":"pith:Q34QCN3K","schema_version":"1.0","canonical_sha256":"86f901376aec9906f1411be181bb69e70044e18bcced93defd8c767de81aa0f2","source":{"kind":"arxiv","id":"2309.17102","version":2},"attestation_state":"computed","paper":{"title":"Guiding Instruction-based Image Editing via Multimodal Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Tsu-Jui Fu, Wenze Hu, William Yang Wang, Xianzhi Du, Yinfei Yang, Zhe Gan","submitted_at":"2023-09-29T10:01:50Z","abstract_excerpt":"Instruction-based image editing improves the controllability and flexibility of image manipulation via natural commands without elaborate descriptions or regional masks. However, human instructions are sometimes too brief for current methods to capture and follow. Multimodal large language models (MLLMs) show promising capabilities in cross-modal understanding and visual-aware response generation via LMs. We investigate how MLLMs facilitate edit instructions and present MLLM-Guided Image Editing (MGIE). MGIE learns to derive expressive instructions and provides explicit guidance. The editing m"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.17102","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-09-29T10:01:50Z","cross_cats_sorted":[],"title_canon_sha256":"554ed35fdd4b839a26c3e0dc7637b1fda9af7a7445939fc493e08d627befea42","abstract_canon_sha256":"ee9790f85fbbd9820aea9145d77703b56271ac8f64c364802760fb742b2a8b7a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:40:57.595804Z","signature_b64":"eIU4VGfpnA1R0TXstq9IhToatRCvMXFKIaGgn5XQkrmlenyOFGMM59iYn5yrSOCqbAkfZ3twFGbT9jhQ1mbyCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"86f901376aec9906f1411be181bb69e70044e18bcced93defd8c767de81aa0f2","last_reissued_at":"2026-07-05T07:40:57.595254Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:40:57.595254Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Guiding Instruction-based Image Editing via Multimodal Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Tsu-Jui Fu, Wenze Hu, William Yang Wang, Xianzhi Du, Yinfei Yang, Zhe Gan","submitted_at":"2023-09-29T10:01:50Z","abstract_excerpt":"Instruction-based image editing improves the controllability and flexibility of image manipulation via natural commands without elaborate descriptions or regional masks. However, human instructions are sometimes too brief for current methods to capture and follow. Multimodal large language models (MLLMs) show promising capabilities in cross-modal understanding and visual-aware response generation via LMs. We investigate how MLLMs facilitate edit instructions and present MLLM-Guided Image Editing (MGIE). MGIE learns to derive expressive instructions and provides explicit guidance. The editing m"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.17102","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.17102/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.17102","created_at":"2026-07-05T07:40:57.595325+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.17102v2","created_at":"2026-07-05T07:40:57.595325+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.17102","created_at":"2026-07-05T07:40:57.595325+00:00"},{"alias_kind":"pith_short_12","alias_value":"Q34QCN3K5SMQ","created_at":"2026-07-05T07:40:57.595325+00:00"},{"alias_kind":"pith_short_16","alias_value":"Q34QCN3K5SMQN4KB","created_at":"2026-07-05T07:40:57.595325+00:00"},{"alias_kind":"pith_short_8","alias_value":"Q34QCN3K","created_at":"2026-07-05T07:40:57.595325+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":20,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26872","citing_title":"SpatialFlow-GRPO: Where Spatial Credit Drives Image Editing","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05142","citing_title":"GeM-NR: Geometry-Aware Multi-View Editing for Nonrigid Scene Changes","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05071","citing_title":"InstantRetouch: Efficient and High-Fidelity Instruction-Guided Image Retouching with Bilateral Space","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14842","citing_title":"Editor's Choice: Evaluating Abstract Intent in Image Editing through Atomic Entity Analysis","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09233","citing_title":"Towards Robust Sequential Decomposition for Complex Image Editing","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26872","citing_title":"SpatialFlow-GRPO: Where Spatial Credit Drives Image Editing","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23518","citing_title":"VINS-120K: Ultra High-Resolution Image Editing with A Large-Scale Dataset","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2602.23622","citing_title":"DLEBench: Evaluating Small-scale Object Editing Ability for Instruction-based Image Editing Model","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2504.20690","citing_title":"In-Context Edit: Enabling Instructional Image Editing with In-Context Generation in Large Scale Diffusion Transformer","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2403.09611","citing_title":"MM1: Methods, Analysis & Insights from Multimodal LLM Pre-training","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15181","citing_title":"From Plans to Pixels: Learning to Plan and Orchestrate for Open-Ended Image Editing","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13047","citing_title":"Revealing the Gap in Human and VLM Scene Perception through Counterfactual Semantic Saliency","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03156","citing_title":"CAMEO: A Conditional and Quality-Aware Multi-Agent Image Editing Orchestrator","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2505.20275","citing_title":"ImgEdit: A Unified Image Editing Dataset and Benchmark","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08250","citing_title":"Why Do DiT Editors Drift? Plug-and-Play Low Frequency Alignment in VAE Latent Space","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09233","citing_title":"Towards Robust Sequential Decomposition for Complex Image Editing","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10454","citing_title":"AIM-Bench: Benchmarking and Improving Affective Image Manipulation via Fine-Grained Hierarchical Control","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08213","citing_title":"EditCaption: Human-Refined SFT and HAE-DPO for Image Editing Instruction Synthesis","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15917","citing_title":"Making Image Editing Easier via Adaptive Task Reformulation with Agentic Executions","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20258","citing_title":"Rethinking Where to Edit: Task-Aware Localization for Instruction-Based Image Editing","ref_index":11,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Q34QCN3K5SMQN4KBDPQYDO3J44","json":"https://pith.science/pith/Q34QCN3K5SMQN4KBDPQYDO3J44.json","graph_json":"https://pith.science/api/pith-number/Q34QCN3K5SMQN4KBDPQYDO3J44/graph.json","events_json":"https://pith.science/api/pith-number/Q34QCN3K5SMQN4KBDPQYDO3J44/events.json","paper":"https://pith.science/paper/Q34QCN3K"},"agent_actions":{"view_html":"https://pith.science/pith/Q34QCN3K5SMQN4KBDPQYDO3J44","download_json":"https://pith.science/pith/Q34QCN3K5SMQN4KBDPQYDO3J44.json","view_paper":"https://pith.science/paper/Q34QCN3K","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.17102&json=true","fetch_graph":"https://pith.science/api/pith-number/Q34QCN3K5SMQN4KBDPQYDO3J44/graph.json","fetch_events":"https://pith.science/api/pith-number/Q34QCN3K5SMQN4KBDPQYDO3J44/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Q34QCN3K5SMQN4KBDPQYDO3J44/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Q34QCN3K5SMQN4KBDPQYDO3J44/action/storage_attestation","attest_author":"https://pith.science/pith/Q34QCN3K5SMQN4KBDPQYDO3J44/action/author_attestation","sign_citation":"https://pith.science/pith/Q34QCN3K5SMQN4KBDPQYDO3J44/action/citation_signature","submit_replication":"https://pith.science/pith/Q34QCN3K5SMQN4KBDPQYDO3J44/action/replication_record"}},"created_at":"2026-07-05T07:40:57.595325+00:00","updated_at":"2026-07-05T07:40:57.595325+00:00"}