{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:PXHF6CJGLYV5SBEPVV43QDAIBP","short_pith_number":"pith:PXHF6CJG","schema_version":"1.0","canonical_sha256":"7dce5f09265e2bd9048fad79b80c080bdc3634e2a61b6b0201d5402f253faf3e","source":{"kind":"arxiv","id":"2210.04885","version":5},"attestation_state":"computed","paper":{"title":"What the DAAM: Interpreting Stable Diffusion Using Cross Attention","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Akshat Pandey, Ferhan Ture, Gefei Yang, Jimmy Lin, Karun Kumar, Linqing Liu, Pontus Stenetorp, Raphael Tang, Zhiying Jiang","submitted_at":"2022-10-10T17:55:41Z","abstract_excerpt":"Large-scale diffusion neural networks represent a substantial milestone in text-to-image generation, but they remain poorly understood, lacking interpretability analyses. In this paper, we perform a text-image attribution analysis on Stable Diffusion, a recently open-sourced model. To produce pixel-level attribution maps, we upscale and aggregate cross-attention word-pixel scores in the denoising subnetwork, naming our method DAAM. We evaluate its correctness by testing its semantic segmentation ability on nouns, as well as its generalized attribution quality on all parts of speech, rated by h"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2210.04885","kind":"arxiv","version":5},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2022-10-10T17:55:41Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"6cbcf81d454cb009812de480380f766dd6fff0bdaf8acb0ceab715c85546ba08","abstract_canon_sha256":"6d3fab83f32593445968af088bf53c8ba101ce78b0718bac2440fd909701f216"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:23:33.814567Z","signature_b64":"MmI+mzL7iYiIvWIWFOk5NO5/cml5l0hqrWagZpjj884jwV4CP8F3bSfF5W0EYT7c78pMHrFRmfsUpArCVsTDDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7dce5f09265e2bd9048fad79b80c080bdc3634e2a61b6b0201d5402f253faf3e","last_reissued_at":"2026-07-05T05:23:33.814078Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:23:33.814078Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"What the DAAM: Interpreting Stable Diffusion Using Cross Attention","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Akshat Pandey, Ferhan Ture, Gefei Yang, Jimmy Lin, Karun Kumar, Linqing Liu, Pontus Stenetorp, Raphael Tang, Zhiying Jiang","submitted_at":"2022-10-10T17:55:41Z","abstract_excerpt":"Large-scale diffusion neural networks represent a substantial milestone in text-to-image generation, but they remain poorly understood, lacking interpretability analyses. In this paper, we perform a text-image attribution analysis on Stable Diffusion, a recently open-sourced model. To produce pixel-level attribution maps, we upscale and aggregate cross-attention word-pixel scores in the denoising subnetwork, naming our method DAAM. We evaluate its correctness by testing its semantic segmentation ability on nouns, as well as its generalized attribution quality on all parts of speech, rated by h"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2210.04885","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2210.04885/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2210.04885","created_at":"2026-07-05T05:23:33.814136+00:00"},{"alias_kind":"arxiv_version","alias_value":"2210.04885v5","created_at":"2026-07-05T05:23:33.814136+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2210.04885","created_at":"2026-07-05T05:23:33.814136+00:00"},{"alias_kind":"pith_short_12","alias_value":"PXHF6CJGLYV5","created_at":"2026-07-05T05:23:33.814136+00:00"},{"alias_kind":"pith_short_16","alias_value":"PXHF6CJGLYV5SBEP","created_at":"2026-07-05T05:23:33.814136+00:00"},{"alias_kind":"pith_short_8","alias_value":"PXHF6CJG","created_at":"2026-07-05T05:23:33.814136+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03715","citing_title":"Text-to-Image Models Need Less from Text Encoders Than You Think","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26460","citing_title":"AnchorDiff: Training-Free Concept Grounding for MM-DiTs via Anchor-Based Graph Propagation","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2509.13742","citing_title":"Spatial Balancing: Designing an LLM-Powered Spatial Externalization Interface for Iterative Science Communication Writing","ref_index":86,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17118","citing_title":"Differentiable Optimization Layers for Guaranteed Fairness in Deep Learning","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2505.19261","citing_title":"Enhancing Text-to-Image Diffusion Transformer via Split-Text Conditioning","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2509.04123","citing_title":"TaleDiffusion: Multi-Character Story Generation with Dialogue Rendering","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2601.06338","citing_title":"Circuit Mechanisms for Spatial Relation Generation in Diffusion Transformers","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10019","citing_title":"The two clocks and the innovation window: When and how generative models learn rules","ref_index":71,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23540","citing_title":"Oracle Noise: Faster Semantic Spherical Alignment for Interpretable Latent Optimization","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05906","citing_title":"Selective Aggregation of Attention Maps Improves Diffusion-Based Visual Interpretation","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20936","citing_title":"AttentionBender: Manipulating Cross-Attention in Video Diffusion Transformers as a Creative Probe","ref_index":68,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PXHF6CJGLYV5SBEPVV43QDAIBP","json":"https://pith.science/pith/PXHF6CJGLYV5SBEPVV43QDAIBP.json","graph_json":"https://pith.science/api/pith-number/PXHF6CJGLYV5SBEPVV43QDAIBP/graph.json","events_json":"https://pith.science/api/pith-number/PXHF6CJGLYV5SBEPVV43QDAIBP/events.json","paper":"https://pith.science/paper/PXHF6CJG"},"agent_actions":{"view_html":"https://pith.science/pith/PXHF6CJGLYV5SBEPVV43QDAIBP","download_json":"https://pith.science/pith/PXHF6CJGLYV5SBEPVV43QDAIBP.json","view_paper":"https://pith.science/paper/PXHF6CJG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2210.04885&json=true","fetch_graph":"https://pith.science/api/pith-number/PXHF6CJGLYV5SBEPVV43QDAIBP/graph.json","fetch_events":"https://pith.science/api/pith-number/PXHF6CJGLYV5SBEPVV43QDAIBP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PXHF6CJGLYV5SBEPVV43QDAIBP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PXHF6CJGLYV5SBEPVV43QDAIBP/action/storage_attestation","attest_author":"https://pith.science/pith/PXHF6CJGLYV5SBEPVV43QDAIBP/action/author_attestation","sign_citation":"https://pith.science/pith/PXHF6CJGLYV5SBEPVV43QDAIBP/action/citation_signature","submit_replication":"https://pith.science/pith/PXHF6CJGLYV5SBEPVV43QDAIBP/action/replication_record"}},"created_at":"2026-07-05T05:23:33.814136+00:00","updated_at":"2026-07-05T05:23:33.814136+00:00"}