{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:NTOX36GLR3KMKFQ67C6QQMEEGF","short_pith_number":"pith:NTOX36GL","schema_version":"1.0","canonical_sha256":"6cdd7df8cb8ed4c5161ef8bd083084316691fe2ae36e665819c980d3bb0fb1c4","source":{"kind":"arxiv","id":"2211.09800","version":2},"attestation_state":"computed","paper":{"title":"InstructPix2Pix: Learning to Follow Image Editing Instructions","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.GR","cs.LG"],"primary_cat":"cs.CV","authors_text":"Aleksander Holynski, Alexei A. Efros, Tim Brooks","submitted_at":"2022-11-17T18:58:43Z","abstract_excerpt":"We propose a method for editing images from human instructions: given an input image and a written instruction that tells the model what to do, our model follows these instructions to edit the image. To obtain training data for this problem, we combine the knowledge of two large pretrained models -- a language model (GPT-3) and a text-to-image model (Stable Diffusion) -- to generate a large dataset of image editing examples. Our conditional diffusion model, InstructPix2Pix, is trained on our generated data, and generalizes to real images and user-written instructions at inference time. Since i"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2211.09800","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2022-11-17T18:58:43Z","cross_cats_sorted":["cs.AI","cs.CL","cs.GR","cs.LG"],"title_canon_sha256":"430cdc0f42be13265d7461ad3a5ebc733132ae4110dfe3ec407b6295690e44f2","abstract_canon_sha256":"c2cdab90b880a8210830ac83f1c2f6e6e60aa64032c5983fea85be28aa211d44"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:34:04.246138Z","signature_b64":"7Sx8Z3N8rq/hVZztA4Sm9/cEqZ8QwSXacZ/9x5Q6FJEscscMwQxsNsehkFGp5LpvuKt+csqtlxzb1UjkrLFxDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6cdd7df8cb8ed4c5161ef8bd083084316691fe2ae36e665819c980d3bb0fb1c4","last_reissued_at":"2026-07-05T05:34:04.245601Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:34:04.245601Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"InstructPix2Pix: Learning to Follow Image Editing Instructions","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.GR","cs.LG"],"primary_cat":"cs.CV","authors_text":"Aleksander Holynski, Alexei A. Efros, Tim Brooks","submitted_at":"2022-11-17T18:58:43Z","abstract_excerpt":"We propose a method for editing images from human instructions: given an input image and a written instruction that tells the model what to do, our model follows these instructions to edit the image. To obtain training data for this problem, we combine the knowledge of two large pretrained models -- a language model (GPT-3) and a text-to-image model (Stable Diffusion) -- to generate a large dataset of image editing examples. Our conditional diffusion model, InstructPix2Pix, is trained on our generated data, and generalizes to real images and user-written instructions at inference time. Since i"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2211.09800","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2211.09800/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2211.09800","created_at":"2026-07-05T05:34:04.245671+00:00"},{"alias_kind":"arxiv_version","alias_value":"2211.09800v2","created_at":"2026-07-05T05:34:04.245671+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2211.09800","created_at":"2026-07-05T05:34:04.245671+00:00"},{"alias_kind":"pith_short_12","alias_value":"NTOX36GLR3KM","created_at":"2026-07-05T05:34:04.245671+00:00"},{"alias_kind":"pith_short_16","alias_value":"NTOX36GLR3KMKFQ6","created_at":"2026-07-05T05:34:04.245671+00:00"},{"alias_kind":"pith_short_8","alias_value":"NTOX36GL","created_at":"2026-07-05T05:34:04.245671+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":24,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23221","citing_title":"RS-Gen: A Multi-Stage Agentic Framework for Reasoning and Search-Augmented Image Generation","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01825","citing_title":"ROGLE: Robust Global-Local Alignment with Automated Region Supervision for Text-Based Person Search","ref_index":135,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05665","citing_title":"V2V-Bench: A Comprehensive Benchmark for Video-to-Video Generation Evaluation","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04306","citing_title":"Organizational Control Layer: Governance Infrastructure at the Execution Boundary of LLM Agent Systems","ref_index":114,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03401","citing_title":"Towards Characterizing Scientific Image Utility and Upgradability","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01825","citing_title":"ROGLE: Robust Global-Local Alignment with Automated Region Supervision for Text-Based Person Search","ref_index":135,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25343","citing_title":"Toward Native Multimodal Modeling: A Roadmap","ref_index":102,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07568","citing_title":"A Systematic Study of Behavioral Cloning for Scientific Data Annotation","ref_index":252,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23245","citing_title":"SimInsert: Seamless Video Object Insertion via Regional Sparse Attention Fusion","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2501.12202","citing_title":"Hunyuan3D 2.0: Scaling Diffusion Models for High Resolution Textured 3D Assets Generation","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21611","citing_title":"UniVL: Unified Vision-Language Embedding for Spatially Grounded Contextual Image Generation","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18010","citing_title":"Functionalization via Structure Completion and Motion Rectification","ref_index":103,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16951","citing_title":"Edit-GRPO: A Locality-Preserving Policy Optimization Framework for Image Editing","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2507.22163","citing_title":"IdeaBlocks: Expressing and Reusing Divergent Intents for Graphic Design Exploration using Generative AI","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2509.05342","citing_title":"Delta Rectified Flow Sampling for Text-to-Image Editing","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2302.11550","citing_title":"Scaling Robot Learning with Semantically Imagined Experience","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2302.05543","citing_title":"Adding Conditional Control to Text-to-Image Diffusion Models","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2603.14186","citing_title":"Setting-Matched and Semantics-Scaled Benchmarking of One-Step Generative Models Against Multistep Diffusion and Flow Models","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2303.04671","citing_title":"Visual ChatGPT: Talking, Drawing and Editing with Visual Foundation Models","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04974","citing_title":"From Video to Control: A Survey of Learning Manipulation Interfaces from Temporal Visual Data","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00707","citing_title":"PhysEdit: Physically-Consistent Region-Aware Image Editing via Adaptive Spatio-Temporal Reasoning","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2304.08485","citing_title":"Visual Instruction Tuning","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13863","citing_title":"PostureObjectstitch: Anomaly Image Generation Considering Assembly Relationships in Industrial Scenarios","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02583","citing_title":"Stylistic Attribute Control in Latent Diffusion Models","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NTOX36GLR3KMKFQ67C6QQMEEGF","json":"https://pith.science/pith/NTOX36GLR3KMKFQ67C6QQMEEGF.json","graph_json":"https://pith.science/api/pith-number/NTOX36GLR3KMKFQ67C6QQMEEGF/graph.json","events_json":"https://pith.science/api/pith-number/NTOX36GLR3KMKFQ67C6QQMEEGF/events.json","paper":"https://pith.science/paper/NTOX36GL"},"agent_actions":{"view_html":"https://pith.science/pith/NTOX36GLR3KMKFQ67C6QQMEEGF","download_json":"https://pith.science/pith/NTOX36GLR3KMKFQ67C6QQMEEGF.json","view_paper":"https://pith.science/paper/NTOX36GL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2211.09800&json=true","fetch_graph":"https://pith.science/api/pith-number/NTOX36GLR3KMKFQ67C6QQMEEGF/graph.json","fetch_events":"https://pith.science/api/pith-number/NTOX36GLR3KMKFQ67C6QQMEEGF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NTOX36GLR3KMKFQ67C6QQMEEGF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NTOX36GLR3KMKFQ67C6QQMEEGF/action/storage_attestation","attest_author":"https://pith.science/pith/NTOX36GLR3KMKFQ67C6QQMEEGF/action/author_attestation","sign_citation":"https://pith.science/pith/NTOX36GLR3KMKFQ67C6QQMEEGF/action/citation_signature","submit_replication":"https://pith.science/pith/NTOX36GLR3KMKFQ67C6QQMEEGF/action/replication_record"}},"created_at":"2026-07-05T05:34:04.245671+00:00","updated_at":"2026-07-05T05:34:04.245671+00:00"}