{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:K5W5IRX2VAMAXFEWQTW73FMTCJ","short_pith_number":"pith:K5W5IRX2","schema_version":"1.0","canonical_sha256":"576dd446faa8180b949684edfd95931254fb03c97e023185a967edc225b89a1d","source":{"kind":"arxiv","id":"2005.00928","version":2},"attestation_state":"computed","paper":{"title":"Quantifying Attention Flow in Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Samira Abnar, Willem Zuidema","submitted_at":"2020-05-02T21:45:27Z","abstract_excerpt":"In the Transformer model, \"self-attention\" combines information from attended embeddings into the representation of the focal embedding in the next layer. Thus, across layers of the Transformer, information originating from different tokens gets increasingly mixed. This makes attention weights unreliable as explanations probes. In this paper, we consider the problem of quantifying this flow of information through self-attention. We propose two methods for approximating the attention to input tokens given attention weights, attention rollout and attention flow, as post hoc methods when we use a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2005.00928","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-05-02T21:45:27Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"a9e5bf5405ff9c17a90407ea058ebdd5d8f8776a141cfbb4f4ed0caa629bc952","abstract_canon_sha256":"b09ab32a9c88a03104ba0269403b722771f24aed4cc971959ea5f79eb883d485"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:06:55.007353Z","signature_b64":"gvGc5eG23fpdtotdbfSks+Rur182oAghr+o1cN8/STchoai41+5VmZWmlLimdbY9Se6gANFxv8yK1x+JEBcVDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"576dd446faa8180b949684edfd95931254fb03c97e023185a967edc225b89a1d","last_reissued_at":"2026-07-05T01:06:55.006803Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:06:55.006803Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Quantifying Attention Flow in Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Samira Abnar, Willem Zuidema","submitted_at":"2020-05-02T21:45:27Z","abstract_excerpt":"In the Transformer model, \"self-attention\" combines information from attended embeddings into the representation of the focal embedding in the next layer. Thus, across layers of the Transformer, information originating from different tokens gets increasingly mixed. This makes attention weights unreliable as explanations probes. In this paper, we consider the problem of quantifying this flow of information through self-attention. We propose two methods for approximating the attention to input tokens given attention weights, attention rollout and attention flow, as post hoc methods when we use a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2005.00928","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2005.00928/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2005.00928","created_at":"2026-07-05T01:06:55.006863+00:00"},{"alias_kind":"arxiv_version","alias_value":"2005.00928v2","created_at":"2026-07-05T01:06:55.006863+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2005.00928","created_at":"2026-07-05T01:06:55.006863+00:00"},{"alias_kind":"pith_short_12","alias_value":"K5W5IRX2VAMA","created_at":"2026-07-05T01:06:55.006863+00:00"},{"alias_kind":"pith_short_16","alias_value":"K5W5IRX2VAMAXFEW","created_at":"2026-07-05T01:06:55.006863+00:00"},{"alias_kind":"pith_short_8","alias_value":"K5W5IRX2","created_at":"2026-07-05T01:06:55.006863+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":19,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24937","citing_title":"The Hitchhiker's Guide to Agentic AI: From Foundations to Systems","ref_index":78,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03569","citing_title":"When Attention Collapses: Stage-Aware Visual Token Pruning from Structure to Semantics","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14255","citing_title":"Architecture-Aware Explanation Auditing for Industrial Visual Inspection","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26190","citing_title":"HRVConformer: Neonatal Hypoxic-Ischemic Encephalopathy Classification from the Heart Rate signals","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26460","citing_title":"AnchorDiff: Training-Free Concept Grounding for MM-DiTs via Anchor-Based Graph Propagation","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07604","citing_title":"Contribution Weights: A Geometrical Analysis of Self-Attention Transformers","ref_index":70,"is_internal_anchor":false},{"citing_arxiv_id":"2409.10102","citing_title":"Trustworthiness in Retrieval-Augmented Generation Systems: A Survey","ref_index":122,"is_internal_anchor":false},{"citing_arxiv_id":"2504.05454","citing_title":"GraphPINE: Graph Importance Propagation for Interpretable Drug Response Prediction","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2602.16608","citing_title":"Explainable AI: Context-Aware Layer-Wise Integrated Gradients for Explaining Transformer Models","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14255","citing_title":"Architecture-Aware Explanation Auditing for Industrial Visual Inspection","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18681","citing_title":"Learning Quantifiable Visual Explanations Without Ground-Truth","ref_index":71,"is_internal_anchor":false},{"citing_arxiv_id":"2508.04427","citing_title":"Decoding the Multimodal Maze: A Systematic Review on the Adoption of Explainability in Multimodal Attention-based Models","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14255","citing_title":"Architecture-Aware Explanation Auditing for Industrial Visual Inspection","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11885","citing_title":"From Clever Hans to Scientific Discovery: Interpreting EEG Foundational Transformers with LRP","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2205.06175","citing_title":"A Generalist Agent","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09271","citing_title":"Shaping Schema via Language Representation as the Next Frontier for LLM Intelligence Expanding","ref_index":97,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13258","citing_title":"Hessian-Enhanced Token Attribution (HETA): Interpreting Autoregressive LLMs","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16485","citing_title":"Saccade Attention Networks: Using Transfer Learning of Attention to Reduce Network Sizes","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02707","citing_title":"SAIL: Structure-Aware Interpretable Learning for Anatomy-Aligned Post-hoc Explanations in OCT","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/K5W5IRX2VAMAXFEWQTW73FMTCJ","json":"https://pith.science/pith/K5W5IRX2VAMAXFEWQTW73FMTCJ.json","graph_json":"https://pith.science/api/pith-number/K5W5IRX2VAMAXFEWQTW73FMTCJ/graph.json","events_json":"https://pith.science/api/pith-number/K5W5IRX2VAMAXFEWQTW73FMTCJ/events.json","paper":"https://pith.science/paper/K5W5IRX2"},"agent_actions":{"view_html":"https://pith.science/pith/K5W5IRX2VAMAXFEWQTW73FMTCJ","download_json":"https://pith.science/pith/K5W5IRX2VAMAXFEWQTW73FMTCJ.json","view_paper":"https://pith.science/paper/K5W5IRX2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2005.00928&json=true","fetch_graph":"https://pith.science/api/pith-number/K5W5IRX2VAMAXFEWQTW73FMTCJ/graph.json","fetch_events":"https://pith.science/api/pith-number/K5W5IRX2VAMAXFEWQTW73FMTCJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/K5W5IRX2VAMAXFEWQTW73FMTCJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/K5W5IRX2VAMAXFEWQTW73FMTCJ/action/storage_attestation","attest_author":"https://pith.science/pith/K5W5IRX2VAMAXFEWQTW73FMTCJ/action/author_attestation","sign_citation":"https://pith.science/pith/K5W5IRX2VAMAXFEWQTW73FMTCJ/action/citation_signature","submit_replication":"https://pith.science/pith/K5W5IRX2VAMAXFEWQTW73FMTCJ/action/replication_record"}},"created_at":"2026-07-05T01:06:55.006863+00:00","updated_at":"2026-07-05T01:06:55.006863+00:00"}