{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:54DQJBWGGM5EHWJZVUXUVAPBRS","short_pith_number":"pith:54DQJBWG","schema_version":"1.0","canonical_sha256":"ef070486c6333a43d939ad2f4a81e18c97a6eabb04d9214485b366b6ebc16a46","source":{"kind":"arxiv","id":"2403.03431","version":1},"attestation_state":"computed","paper":{"title":"Towards Understanding Cross and Self-Attention in Stable Diffusion for Text-Guided Image Editing","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bingyan Liu, Chengyu Wang, Jun Huang, Kui Jia, Tingfeng Cao","submitted_at":"2024-03-06T03:32:56Z","abstract_excerpt":"Deep Text-to-Image Synthesis (TIS) models such as Stable Diffusion have recently gained significant popularity for creative Text-to-image generation. Yet, for domain-specific scenarios, tuning-free Text-guided Image Editing (TIE) is of greater importance for application developers, which modify objects or object properties in images by manipulating feature components in attention layers during the generation process. However, little is known about what semantic meanings these attention layers have learned and which parts of the attention maps contribute to the success of image editing. In this"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.03431","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-03-06T03:32:56Z","cross_cats_sorted":[],"title_canon_sha256":"cad79b312ae0c045d7712eabd017b5a050a6a5f9c2a2041974c549c68cab34b5","abstract_canon_sha256":"3e484f95380095f17aab6fc587e83e9833c00c6e9f8c6e5b9bd34d8575c91c64"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:52:48.148498Z","signature_b64":"44LKMg3FkFXqBAlBnrt+Moshd+6xBWbY1fjNVozA5E8yqG3sd/LwwZFrDQ5wFwdbXG29j4T3sn4l/BeUScmiCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ef070486c6333a43d939ad2f4a81e18c97a6eabb04d9214485b366b6ebc16a46","last_reissued_at":"2026-07-05T07:52:48.148059Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:52:48.148059Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards Understanding Cross and Self-Attention in Stable Diffusion for Text-Guided Image Editing","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bingyan Liu, Chengyu Wang, Jun Huang, Kui Jia, Tingfeng Cao","submitted_at":"2024-03-06T03:32:56Z","abstract_excerpt":"Deep Text-to-Image Synthesis (TIS) models such as Stable Diffusion have recently gained significant popularity for creative Text-to-image generation. Yet, for domain-specific scenarios, tuning-free Text-guided Image Editing (TIE) is of greater importance for application developers, which modify objects or object properties in images by manipulating feature components in attention layers during the generation process. However, little is known about what semantic meanings these attention layers have learned and which parts of the attention maps contribute to the success of image editing. In this"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.03431","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.03431/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.03431","created_at":"2026-07-05T07:52:48.148116+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.03431v1","created_at":"2026-07-05T07:52:48.148116+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.03431","created_at":"2026-07-05T07:52:48.148116+00:00"},{"alias_kind":"pith_short_12","alias_value":"54DQJBWGGM5E","created_at":"2026-07-05T07:52:48.148116+00:00"},{"alias_kind":"pith_short_16","alias_value":"54DQJBWGGM5EHWJZ","created_at":"2026-07-05T07:52:48.148116+00:00"},{"alias_kind":"pith_short_8","alias_value":"54DQJBWG","created_at":"2026-07-05T07:52:48.148116+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.10046","citing_title":"Inside the Latent Flow: Causal Deciphering of Attention Dynamics in Audio Separation Foundation Models","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2601.06338","citing_title":"Circuit Mechanisms for Spatial Relation Generation in Diffusion Transformers","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10019","citing_title":"The two clocks and the innovation window: When and how generative models learn rules","ref_index":89,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/54DQJBWGGM5EHWJZVUXUVAPBRS","json":"https://pith.science/pith/54DQJBWGGM5EHWJZVUXUVAPBRS.json","graph_json":"https://pith.science/api/pith-number/54DQJBWGGM5EHWJZVUXUVAPBRS/graph.json","events_json":"https://pith.science/api/pith-number/54DQJBWGGM5EHWJZVUXUVAPBRS/events.json","paper":"https://pith.science/paper/54DQJBWG"},"agent_actions":{"view_html":"https://pith.science/pith/54DQJBWGGM5EHWJZVUXUVAPBRS","download_json":"https://pith.science/pith/54DQJBWGGM5EHWJZVUXUVAPBRS.json","view_paper":"https://pith.science/paper/54DQJBWG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.03431&json=true","fetch_graph":"https://pith.science/api/pith-number/54DQJBWGGM5EHWJZVUXUVAPBRS/graph.json","fetch_events":"https://pith.science/api/pith-number/54DQJBWGGM5EHWJZVUXUVAPBRS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/54DQJBWGGM5EHWJZVUXUVAPBRS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/54DQJBWGGM5EHWJZVUXUVAPBRS/action/storage_attestation","attest_author":"https://pith.science/pith/54DQJBWGGM5EHWJZVUXUVAPBRS/action/author_attestation","sign_citation":"https://pith.science/pith/54DQJBWGGM5EHWJZVUXUVAPBRS/action/citation_signature","submit_replication":"https://pith.science/pith/54DQJBWGGM5EHWJZVUXUVAPBRS/action/replication_record"}},"created_at":"2026-07-05T07:52:48.148116+00:00","updated_at":"2026-07-05T07:52:48.148116+00:00"}