{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:COHIU7Y7JXT4DX723HHGICHUUC","short_pith_number":"pith:COHIU7Y7","schema_version":"1.0","canonical_sha256":"138e8a7f1f4de7c1dffad9ce6408f4a08b63e9abcd767c1ae293e5bc3a3b31c1","source":{"kind":"arxiv","id":"2411.15034","version":1},"attestation_state":"computed","paper":{"title":"HeadRouter: A Training-free Image Editing Framework for MM-DiTs by Adaptively Routing Attention Heads","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Fan Tang, Jintao Li, Juan Cao, Oliver Deussen, Tong-Yee Lee, Xiaoyu Kong, Yuxin Zhang, Yu Xu","submitted_at":"2024-11-22T16:08:03Z","abstract_excerpt":"Diffusion Transformers (DiTs) have exhibited robust capabilities in image generation tasks. However, accurate text-guided image editing for multimodal DiTs (MM-DiTs) still poses a significant challenge. Unlike UNet-based structures that could utilize self/cross-attention maps for semantic editing, MM-DiTs inherently lack support for explicit and consistent incorporated text guidance, resulting in semantic misalignment between the edited results and texts. In this study, we disclose the sensitivity of different attention heads to different image semantics within MM-DiTs and introduce HeadRouter"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.15034","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-11-22T16:08:03Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"540d266a6ff98fffc78b7dc54f3204304ce0d530c75215af507ca4b0beaf3fe6","abstract_canon_sha256":"e02430bb37c920b9e13e4372c8c4a01a2888927642a5dcde8d7492612784d350"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:39:15.385592Z","signature_b64":"pOlyi+Xyy7o4PFRI9hl7xy828jkNnSJLkOUqA9gvKbv11A7tuGMIozrH9q8AV22DgttJcTzmazqiJfMhs8eGDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"138e8a7f1f4de7c1dffad9ce6408f4a08b63e9abcd767c1ae293e5bc3a3b31c1","last_reissued_at":"2026-07-05T09:39:15.385165Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:39:15.385165Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HeadRouter: A Training-free Image Editing Framework for MM-DiTs by Adaptively Routing Attention Heads","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Fan Tang, Jintao Li, Juan Cao, Oliver Deussen, Tong-Yee Lee, Xiaoyu Kong, Yuxin Zhang, Yu Xu","submitted_at":"2024-11-22T16:08:03Z","abstract_excerpt":"Diffusion Transformers (DiTs) have exhibited robust capabilities in image generation tasks. However, accurate text-guided image editing for multimodal DiTs (MM-DiTs) still poses a significant challenge. Unlike UNet-based structures that could utilize self/cross-attention maps for semantic editing, MM-DiTs inherently lack support for explicit and consistent incorporated text guidance, resulting in semantic misalignment between the edited results and texts. In this study, we disclose the sensitivity of different attention heads to different image semantics within MM-DiTs and introduce HeadRouter"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.15034","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.15034/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.15034","created_at":"2026-07-05T09:39:15.385226+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.15034v1","created_at":"2026-07-05T09:39:15.385226+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.15034","created_at":"2026-07-05T09:39:15.385226+00:00"},{"alias_kind":"pith_short_12","alias_value":"COHIU7Y7JXT4","created_at":"2026-07-05T09:39:15.385226+00:00"},{"alias_kind":"pith_short_16","alias_value":"COHIU7Y7JXT4DX72","created_at":"2026-07-05T09:39:15.385226+00:00"},{"alias_kind":"pith_short_8","alias_value":"COHIU7Y7","created_at":"2026-07-05T09:39:15.385226+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2504.13109","citing_title":"UniEdit-Flow: Unleashing Inversion and Editing in the Era of Flow Models","ref_index":80,"is_internal_anchor":false},{"citing_arxiv_id":"2509.22244","citing_title":"FlashEdit: Decoupling Speed, Structure, and Semantics for Precise Image Editing","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24625","citing_title":"Meta-CoT: Enhancing Granularity and Generalization in Image Editing","ref_index":68,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/COHIU7Y7JXT4DX723HHGICHUUC","json":"https://pith.science/pith/COHIU7Y7JXT4DX723HHGICHUUC.json","graph_json":"https://pith.science/api/pith-number/COHIU7Y7JXT4DX723HHGICHUUC/graph.json","events_json":"https://pith.science/api/pith-number/COHIU7Y7JXT4DX723HHGICHUUC/events.json","paper":"https://pith.science/paper/COHIU7Y7"},"agent_actions":{"view_html":"https://pith.science/pith/COHIU7Y7JXT4DX723HHGICHUUC","download_json":"https://pith.science/pith/COHIU7Y7JXT4DX723HHGICHUUC.json","view_paper":"https://pith.science/paper/COHIU7Y7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.15034&json=true","fetch_graph":"https://pith.science/api/pith-number/COHIU7Y7JXT4DX723HHGICHUUC/graph.json","fetch_events":"https://pith.science/api/pith-number/COHIU7Y7JXT4DX723HHGICHUUC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/COHIU7Y7JXT4DX723HHGICHUUC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/COHIU7Y7JXT4DX723HHGICHUUC/action/storage_attestation","attest_author":"https://pith.science/pith/COHIU7Y7JXT4DX723HHGICHUUC/action/author_attestation","sign_citation":"https://pith.science/pith/COHIU7Y7JXT4DX723HHGICHUUC/action/citation_signature","submit_replication":"https://pith.science/pith/COHIU7Y7JXT4DX723HHGICHUUC/action/replication_record"}},"created_at":"2026-07-05T09:39:15.385226+00:00","updated_at":"2026-07-05T09:39:15.385226+00:00"}