{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:FIJZQA5I2QKEEJZ7RILASMJY7X","short_pith_number":"pith:FIJZQA5I","schema_version":"1.0","canonical_sha256":"2a139803a8d41442273f8a16093138fdec99b490c6dc3bd2cde2b5f7b7ddf466","source":{"kind":"arxiv","id":"2504.02826","version":4},"attestation_state":"computed","paper":{"title":"Envisioning Beyond the Pixels: Benchmarking Reasoning-Informed Visual Editing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Guangtao Zhai, Haodong Duan, Hao Li, Hua Yang, Junchi Yan, Kexian Tang, Peiyuan Zhang, Renqiu Xia, Wenhao Chai, Xiangyu Zhao, Xiaorong Zhu, Xue Yang, Zicheng Zhang","submitted_at":"2025-04-03T17:59:56Z","abstract_excerpt":"Large Multi-modality Models (LMMs) have made significant progress in visual understanding and generation, but they still face challenges in General Visual Editing, particularly in following complex instructions, preserving appearance consistency, and supporting flexible input formats. To study this gap, we introduce RISEBench, the first benchmark for evaluating Reasoning-Informed viSual Editing (RISE). RISEBench focuses on four key reasoning categories: Temporal, Causal, Spatial, and Logical Reasoning. We curate high-quality test cases for each category and propose an robust evaluation framewo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.02826","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-04-03T17:59:56Z","cross_cats_sorted":[],"title_canon_sha256":"c667affce1ddc359e797679925d49181af165186619af24f042d01b11d713996","abstract_canon_sha256":"bb24f320353c25fb397234e741a71f058aff35b219fe07f5e95c113e789de2c2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:10:36.867046Z","signature_b64":"V3wOKbmsprAYzPHGU6tvGPyrDrzuIrq1Na2NzhKAoyOaZmA5cnFJR/UplDpF1E0JS9uL2SS2V5+zeDd58iPsBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2a139803a8d41442273f8a16093138fdec99b490c6dc3bd2cde2b5f7b7ddf466","last_reissued_at":"2026-07-05T11:10:36.866528Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:10:36.866528Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Envisioning Beyond the Pixels: Benchmarking Reasoning-Informed Visual Editing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Guangtao Zhai, Haodong Duan, Hao Li, Hua Yang, Junchi Yan, Kexian Tang, Peiyuan Zhang, Renqiu Xia, Wenhao Chai, Xiangyu Zhao, Xiaorong Zhu, Xue Yang, Zicheng Zhang","submitted_at":"2025-04-03T17:59:56Z","abstract_excerpt":"Large Multi-modality Models (LMMs) have made significant progress in visual understanding and generation, but they still face challenges in General Visual Editing, particularly in following complex instructions, preserving appearance consistency, and supporting flexible input formats. To study this gap, we introduce RISEBench, the first benchmark for evaluating Reasoning-Informed viSual Editing (RISE). RISEBench focuses on four key reasoning categories: Temporal, Causal, Spatial, and Logical Reasoning. We curate high-quality test cases for each category and propose an robust evaluation framewo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.02826","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.02826/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.02826","created_at":"2026-07-05T11:10:36.866590+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.02826v4","created_at":"2026-07-05T11:10:36.866590+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.02826","created_at":"2026-07-05T11:10:36.866590+00:00"},{"alias_kind":"pith_short_12","alias_value":"FIJZQA5I2QKE","created_at":"2026-07-05T11:10:36.866590+00:00"},{"alias_kind":"pith_short_16","alias_value":"FIJZQA5I2QKEEJZ7","created_at":"2026-07-05T11:10:36.866590+00:00"},{"alias_kind":"pith_short_8","alias_value":"FIJZQA5I","created_at":"2026-07-05T11:10:36.866590+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":19,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26907","citing_title":"Qwen-Image-Agent: Bridging the Context Gap in Real-World Image Generation","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26551","citing_title":"PhyEditBench: A Real-World Multi-Stage Benchmark for Physics-Aware Image Editing","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26738","citing_title":"Do Image Editing Models Understand Lighting?","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02290","citing_title":"DisciplineGen-1M: A Large-Scale Dataset for Multidisciplinary Visual Generation and Editing","ref_index":85,"is_internal_anchor":false},{"citing_arxiv_id":"2606.13679","citing_title":"InterleaveThinker: Reinforcing Agentic Interleaved Generation","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24023","citing_title":"ServImage: An Image Generation and Editing Benchmark from Real-world Commercial Imaging Services","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26907","citing_title":"Qwen-Image-Agent: Bridging the Context Gap in Real-World Image Generation","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26551","citing_title":"PhyEditBench: A Real-World Multi-Stage Benchmark for Physics-Aware Image Editing","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21487","citing_title":"Uni-Edit: Intelligent Editing Is A General Task For Unified Model Tuning","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2602.23622","citing_title":"DLEBench: Evaluating Small-scale Object Editing Ability for Instruction-based Image Editing Model","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21487","citing_title":"Uni-Edit: Intelligent Editing Is A General Task For Unified Model Tuning","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2603.15432","citing_title":"Gym-V: A Unified Vision Environment System for Agentic Vision Research","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13062","citing_title":"Edit-Compass & EditReward-Compass: A Unified Benchmark for Image Editing and Reward Modeling","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12500","citing_title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","ref_index":166,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11541","citing_title":"GeoR-Bench: Evaluating Geoscience Visual Reasoning","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25477","citing_title":"DDA-Thinker: Decoupled Dual-Atomic Reinforcement Learning for Reasoning-Driven Image Editing","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24625","citing_title":"Meta-CoT: Enhancing Granularity and Generalization in Image Editing","ref_index":84,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24023","citing_title":"ServImage: An Image Generation and Editing Benchmark from Real-world Commercial Imaging Services","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2505.14683","citing_title":"Emerging Properties in Unified Multimodal Pretraining","ref_index":104,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FIJZQA5I2QKEEJZ7RILASMJY7X","json":"https://pith.science/pith/FIJZQA5I2QKEEJZ7RILASMJY7X.json","graph_json":"https://pith.science/api/pith-number/FIJZQA5I2QKEEJZ7RILASMJY7X/graph.json","events_json":"https://pith.science/api/pith-number/FIJZQA5I2QKEEJZ7RILASMJY7X/events.json","paper":"https://pith.science/paper/FIJZQA5I"},"agent_actions":{"view_html":"https://pith.science/pith/FIJZQA5I2QKEEJZ7RILASMJY7X","download_json":"https://pith.science/pith/FIJZQA5I2QKEEJZ7RILASMJY7X.json","view_paper":"https://pith.science/paper/FIJZQA5I","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.02826&json=true","fetch_graph":"https://pith.science/api/pith-number/FIJZQA5I2QKEEJZ7RILASMJY7X/graph.json","fetch_events":"https://pith.science/api/pith-number/FIJZQA5I2QKEEJZ7RILASMJY7X/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FIJZQA5I2QKEEJZ7RILASMJY7X/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FIJZQA5I2QKEEJZ7RILASMJY7X/action/storage_attestation","attest_author":"https://pith.science/pith/FIJZQA5I2QKEEJZ7RILASMJY7X/action/author_attestation","sign_citation":"https://pith.science/pith/FIJZQA5I2QKEEJZ7RILASMJY7X/action/citation_signature","submit_replication":"https://pith.science/pith/FIJZQA5I2QKEEJZ7RILASMJY7X/action/replication_record"}},"created_at":"2026-07-05T11:10:36.866590+00:00","updated_at":"2026-07-05T11:10:36.866590+00:00"}