{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:WOB66ORRVZ4PC2KVWUJVWFNXZ4","short_pith_number":"pith:WOB66ORR","schema_version":"1.0","canonical_sha256":"b383ef3a31ae78f16955b5135b15b7cf2f76f4bc2157b7408a9004935384df7e","source":{"kind":"arxiv","id":"2203.04212","version":3},"attestation_state":"computed","paper":{"title":"Measuring the Mixing of Contextual Information in the Transformer","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Gerard I. G\\'allego, Javier Ferrando, Marta R. Costa-juss\\`a","submitted_at":"2022-03-08T17:21:27Z","abstract_excerpt":"The Transformer architecture aggregates input information through the self-attention mechanism, but there is no clear understanding of how this information is mixed across the entire model. Additionally, recent works have demonstrated that attention weights alone are not enough to describe the flow of information. In this paper, we consider the whole attention block -- multi-head attention, residual connection, and layer normalization -- and define a metric to measure token-to-token interactions within each layer. Then, we aggregate layer-wise interpretations to provide input attribution score"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2203.04212","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2022-03-08T17:21:27Z","cross_cats_sorted":[],"title_canon_sha256":"891b2c327f29e588d047772e0497a738c8437609bfee5aba10ee93e0c4c9be5b","abstract_canon_sha256":"f059b3b16b6adce36b7ae2f1c42597378124f463d9a1517e63babeaded763de2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:09:16.203825Z","signature_b64":"PsLbYU4dNVifbLRQ/5AWuZPdsYasKWbZYlocaanUC7aWCbjR+Enddas4tG6oY6jDWmerG4ZcG8x5bf3hI9VDAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b383ef3a31ae78f16955b5135b15b7cf2f76f4bc2157b7408a9004935384df7e","last_reissued_at":"2026-07-05T05:09:16.203407Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:09:16.203407Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Measuring the Mixing of Contextual Information in the Transformer","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Gerard I. G\\'allego, Javier Ferrando, Marta R. Costa-juss\\`a","submitted_at":"2022-03-08T17:21:27Z","abstract_excerpt":"The Transformer architecture aggregates input information through the self-attention mechanism, but there is no clear understanding of how this information is mixed across the entire model. Additionally, recent works have demonstrated that attention weights alone are not enough to describe the flow of information. In this paper, we consider the whole attention block -- multi-head attention, residual connection, and layer normalization -- and define a metric to measure token-to-token interactions within each layer. Then, we aggregate layer-wise interpretations to provide input attribution score"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2203.04212","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2203.04212/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2203.04212","created_at":"2026-07-05T05:09:16.203464+00:00"},{"alias_kind":"arxiv_version","alias_value":"2203.04212v3","created_at":"2026-07-05T05:09:16.203464+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2203.04212","created_at":"2026-07-05T05:09:16.203464+00:00"},{"alias_kind":"pith_short_12","alias_value":"WOB66ORRVZ4P","created_at":"2026-07-05T05:09:16.203464+00:00"},{"alias_kind":"pith_short_16","alias_value":"WOB66ORRVZ4PC2KV","created_at":"2026-07-05T05:09:16.203464+00:00"},{"alias_kind":"pith_short_8","alias_value":"WOB66ORR","created_at":"2026-07-05T05:09:16.203464+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07604","citing_title":"Contribution Weights: A Geometrical Analysis of Self-Attention Transformers","ref_index":157,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07604","citing_title":"Contribution Weights: A Geometrical Analysis of Self-Attention Transformers","ref_index":71,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23640","citing_title":"CachePrune: Privacy-Aware and Fine-Grained KV Cache Sharing for Efficient LLM Inference","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2508.04427","citing_title":"Decoding the Multimodal Maze: A Systematic Review on the Adoption of Explainability in Multimodal Attention-based Models","ref_index":102,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27914","citing_title":"Geometry-Calibrated Conformal Abstention for Language Models","ref_index":42,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WOB66ORRVZ4PC2KVWUJVWFNXZ4","json":"https://pith.science/pith/WOB66ORRVZ4PC2KVWUJVWFNXZ4.json","graph_json":"https://pith.science/api/pith-number/WOB66ORRVZ4PC2KVWUJVWFNXZ4/graph.json","events_json":"https://pith.science/api/pith-number/WOB66ORRVZ4PC2KVWUJVWFNXZ4/events.json","paper":"https://pith.science/paper/WOB66ORR"},"agent_actions":{"view_html":"https://pith.science/pith/WOB66ORRVZ4PC2KVWUJVWFNXZ4","download_json":"https://pith.science/pith/WOB66ORRVZ4PC2KVWUJVWFNXZ4.json","view_paper":"https://pith.science/paper/WOB66ORR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2203.04212&json=true","fetch_graph":"https://pith.science/api/pith-number/WOB66ORRVZ4PC2KVWUJVWFNXZ4/graph.json","fetch_events":"https://pith.science/api/pith-number/WOB66ORRVZ4PC2KVWUJVWFNXZ4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WOB66ORRVZ4PC2KVWUJVWFNXZ4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WOB66ORRVZ4PC2KVWUJVWFNXZ4/action/storage_attestation","attest_author":"https://pith.science/pith/WOB66ORRVZ4PC2KVWUJVWFNXZ4/action/author_attestation","sign_citation":"https://pith.science/pith/WOB66ORRVZ4PC2KVWUJVWFNXZ4/action/citation_signature","submit_replication":"https://pith.science/pith/WOB66ORRVZ4PC2KVWUJVWFNXZ4/action/replication_record"}},"created_at":"2026-07-05T05:09:16.203464+00:00","updated_at":"2026-07-05T05:09:16.203464+00:00"}