{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:CFSF4IVJTSGKDMWC2CDOIS5NNA","short_pith_number":"pith:CFSF4IVJ","schema_version":"1.0","canonical_sha256":"11645e22a99c8ca1b2c2d086e44bad680e6af4114c5cdd8261524f335c856c6c","source":{"kind":"arxiv","id":"2506.03107","version":2},"attestation_state":"computed","paper":{"title":"ByteMorph: Benchmarking Instruction-Guided Image Editing with Non-Rigid Motions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bo Liu, Di Chang, Gordon Wetzstein, Mingdeng Cao, Mohammad Soleymani, Peng Wang, Shengqu Cai, Shijie Zhou, Weilin Huang, Yichun Shi","submitted_at":"2025-06-03T17:39:47Z","abstract_excerpt":"Editing images with instructions to reflect non-rigid motions, camera viewpoint shifts, object deformations, human articulations, and complex interactions, poses a challenging yet underexplored problem in computer vision. Existing approaches and datasets predominantly focus on static scenes or rigid transformations, limiting their capacity to handle expressive edits involving dynamic motion. To address this gap, we introduce ByteMorph, a comprehensive framework for instruction-based image editing with an emphasis on non-rigid motions. ByteMorph comprises a large-scale dataset, ByteMorph-6M, an"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.03107","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-06-03T17:39:47Z","cross_cats_sorted":[],"title_canon_sha256":"4f1e502b45c4a9d0f5cf4d15ed4f4ff1827f511419e168d23e94f940dc3be6bc","abstract_canon_sha256":"f4722e4d86d119dc15451476636fb3558bcef95ea318765fc08975dd8e5ca281"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:19:28.214552Z","signature_b64":"Rhs7hHwscMDh2Z0Z4zJQExVSzGQMs/AlquKv+TNK0dGqOF0F0sVxx0G8wkQnp0ab+4HTo4PBQ50u+Ix85ybTBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"11645e22a99c8ca1b2c2d086e44bad680e6af4114c5cdd8261524f335c856c6c","last_reissued_at":"2026-07-05T11:19:28.214058Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:19:28.214058Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ByteMorph: Benchmarking Instruction-Guided Image Editing with Non-Rigid Motions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bo Liu, Di Chang, Gordon Wetzstein, Mingdeng Cao, Mohammad Soleymani, Peng Wang, Shengqu Cai, Shijie Zhou, Weilin Huang, Yichun Shi","submitted_at":"2025-06-03T17:39:47Z","abstract_excerpt":"Editing images with instructions to reflect non-rigid motions, camera viewpoint shifts, object deformations, human articulations, and complex interactions, poses a challenging yet underexplored problem in computer vision. Existing approaches and datasets predominantly focus on static scenes or rigid transformations, limiting their capacity to handle expressive edits involving dynamic motion. To address this gap, we introduce ByteMorph, a comprehensive framework for instruction-based image editing with an emphasis on non-rigid motions. ByteMorph comprises a large-scale dataset, ByteMorph-6M, an"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.03107","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.03107/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.03107","created_at":"2026-07-05T11:19:28.214119+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.03107v2","created_at":"2026-07-05T11:19:28.214119+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.03107","created_at":"2026-07-05T11:19:28.214119+00:00"},{"alias_kind":"pith_short_12","alias_value":"CFSF4IVJTSGK","created_at":"2026-07-05T11:19:28.214119+00:00"},{"alias_kind":"pith_short_16","alias_value":"CFSF4IVJTSGKDMWC","created_at":"2026-07-05T11:19:28.214119+00:00"},{"alias_kind":"pith_short_8","alias_value":"CFSF4IVJ","created_at":"2026-07-05T11:19:28.214119+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19073","citing_title":"Taming I2V models for Image HOI Editing: A Cognitive Benchmark and Agentic Self-Correcting Framework","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09233","citing_title":"Towards Robust Sequential Decomposition for Complex Image Editing","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27924","citing_title":"SIGMA: Semantic-Difference Instruction-Grounding Mask Annotator for Text-Driven Image Manipulation Localization","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00931","citing_title":"CV-Arena: An Open Benchmark for Instructional Computer Vision Problem Solving with Human-AI Collaborative Preferences","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09233","citing_title":"Towards Robust Sequential Decomposition for Complex Image Editing","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25477","citing_title":"DDA-Thinker: Decoupled Dual-Atomic Reinforcement Learning for Reasoning-Driven Image Editing","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08213","citing_title":"EditCaption: Human-Refined SFT and HAE-DPO for Image Editing Instruction Synthesis","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07230","citing_title":"PhyEdit: Towards Real-World Object Manipulation via Physically-Grounded Image Editing","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17021","citing_title":"LIVE: Leveraging Image Manipulation Priors for Instruction-based Video Editing","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CFSF4IVJTSGKDMWC2CDOIS5NNA","json":"https://pith.science/pith/CFSF4IVJTSGKDMWC2CDOIS5NNA.json","graph_json":"https://pith.science/api/pith-number/CFSF4IVJTSGKDMWC2CDOIS5NNA/graph.json","events_json":"https://pith.science/api/pith-number/CFSF4IVJTSGKDMWC2CDOIS5NNA/events.json","paper":"https://pith.science/paper/CFSF4IVJ"},"agent_actions":{"view_html":"https://pith.science/pith/CFSF4IVJTSGKDMWC2CDOIS5NNA","download_json":"https://pith.science/pith/CFSF4IVJTSGKDMWC2CDOIS5NNA.json","view_paper":"https://pith.science/paper/CFSF4IVJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.03107&json=true","fetch_graph":"https://pith.science/api/pith-number/CFSF4IVJTSGKDMWC2CDOIS5NNA/graph.json","fetch_events":"https://pith.science/api/pith-number/CFSF4IVJTSGKDMWC2CDOIS5NNA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CFSF4IVJTSGKDMWC2CDOIS5NNA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CFSF4IVJTSGKDMWC2CDOIS5NNA/action/storage_attestation","attest_author":"https://pith.science/pith/CFSF4IVJTSGKDMWC2CDOIS5NNA/action/author_attestation","sign_citation":"https://pith.science/pith/CFSF4IVJTSGKDMWC2CDOIS5NNA/action/citation_signature","submit_replication":"https://pith.science/pith/CFSF4IVJTSGKDMWC2CDOIS5NNA/action/replication_record"}},"created_at":"2026-07-05T11:19:28.214119+00:00","updated_at":"2026-07-05T11:19:28.214119+00:00"}