{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:3GLOXRMISJ4NLRGCBDLIVXTH6S","short_pith_number":"pith:3GLOXRMI","schema_version":"1.0","canonical_sha256":"d996ebc5889278d5c4c208d68ade67f4bc9b4162bf4f7f07572bfffd162124fb","source":{"kind":"arxiv","id":"2403.17924","version":3},"attestation_state":"computed","paper":{"title":"AID: Attention Interpolation of Text-to-Image Diffusion","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Angela Yao, Jinghao Wang, Qiyuan He, Ziwei Liu","submitted_at":"2024-03-26T17:57:05Z","abstract_excerpt":"Conditional diffusion models can create unseen images in various settings, aiding image interpolation. Interpolation in latent spaces is well-studied, but interpolation with specific conditions like text or poses is less understood. Simple approaches, such as linear interpolation in the space of conditions, often result in images that lack consistency, smoothness, and fidelity. To that end, we introduce a novel training-free technique named Attention Interpolation via Diffusion (AID). Our key contributions include 1) proposing an inner/outer interpolated attention layer; 2) fusing the interpol"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.17924","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-03-26T17:57:05Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"31d0f46e1d14cfffda8b264f7aec818bed17f47fd19bd8216c68ab1c101e67ad","abstract_canon_sha256":"ac7e2b9fd2e568c9dc6c1ea19497a9d641e00d9ecab26737a36129c137637f06"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:15:55.241765Z","signature_b64":"OuqFlVkBQvSVAYtOY9LKSizgRBvs0WjY/dvAPGPUoOOLiRHlyg/mmTgpBMQ8R7w2GkE54vSbmF7ue99ndVyQBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d996ebc5889278d5c4c208d68ade67f4bc9b4162bf4f7f07572bfffd162124fb","last_reissued_at":"2026-07-05T09:15:55.241172Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:15:55.241172Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AID: Attention Interpolation of Text-to-Image Diffusion","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Angela Yao, Jinghao Wang, Qiyuan He, Ziwei Liu","submitted_at":"2024-03-26T17:57:05Z","abstract_excerpt":"Conditional diffusion models can create unseen images in various settings, aiding image interpolation. Interpolation in latent spaces is well-studied, but interpolation with specific conditions like text or poses is less understood. Simple approaches, such as linear interpolation in the space of conditions, often result in images that lack consistency, smoothness, and fidelity. To that end, we introduce a novel training-free technique named Attention Interpolation via Diffusion (AID). Our key contributions include 1) proposing an inner/outer interpolated attention layer; 2) fusing the interpol"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.17924","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.17924/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.17924","created_at":"2026-07-05T09:15:55.241233+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.17924v3","created_at":"2026-07-05T09:15:55.241233+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.17924","created_at":"2026-07-05T09:15:55.241233+00:00"},{"alias_kind":"pith_short_12","alias_value":"3GLOXRMISJ4N","created_at":"2026-07-05T09:15:55.241233+00:00"},{"alias_kind":"pith_short_16","alias_value":"3GLOXRMISJ4NLRGC","created_at":"2026-07-05T09:15:55.241233+00:00"},{"alias_kind":"pith_short_8","alias_value":"3GLOXRMI","created_at":"2026-07-05T09:15:55.241233+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.30317","citing_title":"VPG: Visual Prefix Guidance for Autoregressive Image and Video Generation","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2603.07561","citing_title":"PureCC: Pure Learning for Text-to-Image Concept Customization","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3GLOXRMISJ4NLRGCBDLIVXTH6S","json":"https://pith.science/pith/3GLOXRMISJ4NLRGCBDLIVXTH6S.json","graph_json":"https://pith.science/api/pith-number/3GLOXRMISJ4NLRGCBDLIVXTH6S/graph.json","events_json":"https://pith.science/api/pith-number/3GLOXRMISJ4NLRGCBDLIVXTH6S/events.json","paper":"https://pith.science/paper/3GLOXRMI"},"agent_actions":{"view_html":"https://pith.science/pith/3GLOXRMISJ4NLRGCBDLIVXTH6S","download_json":"https://pith.science/pith/3GLOXRMISJ4NLRGCBDLIVXTH6S.json","view_paper":"https://pith.science/paper/3GLOXRMI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.17924&json=true","fetch_graph":"https://pith.science/api/pith-number/3GLOXRMISJ4NLRGCBDLIVXTH6S/graph.json","fetch_events":"https://pith.science/api/pith-number/3GLOXRMISJ4NLRGCBDLIVXTH6S/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3GLOXRMISJ4NLRGCBDLIVXTH6S/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3GLOXRMISJ4NLRGCBDLIVXTH6S/action/storage_attestation","attest_author":"https://pith.science/pith/3GLOXRMISJ4NLRGCBDLIVXTH6S/action/author_attestation","sign_citation":"https://pith.science/pith/3GLOXRMISJ4NLRGCBDLIVXTH6S/action/citation_signature","submit_replication":"https://pith.science/pith/3GLOXRMISJ4NLRGCBDLIVXTH6S/action/replication_record"}},"created_at":"2026-07-05T09:15:55.241233+00:00","updated_at":"2026-07-05T09:15:55.241233+00:00"}