{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:ZHAK3FRWS3IADY2L33GNZ3QBIV","short_pith_number":"pith:ZHAK3FRW","schema_version":"1.0","canonical_sha256":"c9c0ad963696d001e34bdeccdcee0145553fdc9e3e923f52d74c221fe2de1825","source":{"kind":"arxiv","id":"1908.11775","version":4},"attestation_state":"computed","paper":{"title":"Transformer Dissection: A Unified Understanding of Transformer's Attention via the Lens of Kernel","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Louis-Philippe Morency, Makoto Yamada, Ruslan Salakhutdinov, Shaojie Bai, Yao-Hung Hubert Tsai","submitted_at":"2019-08-30T15:05:02Z","abstract_excerpt":"Transformer is a powerful architecture that achieves superior performance on various sequence learning tasks, including neural machine translation, language understanding, and sequence prediction. At the core of the Transformer is the attention mechanism, which concurrently processes all inputs in the streams. In this paper, we present a new formulation of attention via the lens of the kernel. To be more precise, we realize that the attention can be seen as applying kernel smoother over the inputs with the kernel scores being the similarities between inputs. This new formulation gives us a bet"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1908.11775","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-08-30T15:05:02Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"48196112950222f0b89538f712478808f36495d674063c7a152f6869db1afd2c","abstract_canon_sha256":"e2866630a74ab10082f6f8c603cbae63b0cd4f0fbb017bbe9aa849e1aa8e70eb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:18:40.014794Z","signature_b64":"yk+GpoTQTgEW/zQEWJuxSUqI4x6l4iRasLN5EnVJ/SSCuy/9f9BCENS/rM9Cl0cHmu1FoiRCZGkBigvo4lqaDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c9c0ad963696d001e34bdeccdcee0145553fdc9e3e923f52d74c221fe2de1825","last_reissued_at":"2026-07-05T00:18:40.014285Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:18:40.014285Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Transformer Dissection: A Unified Understanding of Transformer's Attention via the Lens of Kernel","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Louis-Philippe Morency, Makoto Yamada, Ruslan Salakhutdinov, Shaojie Bai, Yao-Hung Hubert Tsai","submitted_at":"2019-08-30T15:05:02Z","abstract_excerpt":"Transformer is a powerful architecture that achieves superior performance on various sequence learning tasks, including neural machine translation, language understanding, and sequence prediction. At the core of the Transformer is the attention mechanism, which concurrently processes all inputs in the streams. In this paper, we present a new formulation of attention via the lens of the kernel. To be more precise, we realize that the attention can be seen as applying kernel smoother over the inputs with the kernel scores being the similarities between inputs. This new formulation gives us a bet"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1908.11775","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1908.11775/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1908.11775","created_at":"2026-07-05T00:18:40.014352+00:00"},{"alias_kind":"arxiv_version","alias_value":"1908.11775v4","created_at":"2026-07-05T00:18:40.014352+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1908.11775","created_at":"2026-07-05T00:18:40.014352+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZHAK3FRWS3IA","created_at":"2026-07-05T00:18:40.014352+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZHAK3FRWS3IADY2L","created_at":"2026-07-05T00:18:40.014352+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZHAK3FRW","created_at":"2026-07-05T00:18:40.014352+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.20635","citing_title":"The General Theory of Localization Methods","ref_index":93,"is_internal_anchor":false},{"citing_arxiv_id":"2212.08989","citing_title":"Deep learning applied to computational mechanics: A comprehensive review, state of the art, and the classics","ref_index":229,"is_internal_anchor":false},{"citing_arxiv_id":"2506.10959","citing_title":"Understanding In-Context Learning on Structured Manifolds: Bridging Attention to Kernel Methods","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20271","citing_title":"Multi-Head Attention as Ensemble Nadaraya-Watson Estimation: Variance Reduction, Decorrelation, and Optimal Head Diversity","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20635","citing_title":"The General Theory of Localization Methods","ref_index":123,"is_internal_anchor":false},{"citing_arxiv_id":"2603.02727","citing_title":"Gated Differential Linear Attention: A Linear-Time Decoder for High-Fidelity Medical Segmentation","ref_index":34,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZHAK3FRWS3IADY2L33GNZ3QBIV","json":"https://pith.science/pith/ZHAK3FRWS3IADY2L33GNZ3QBIV.json","graph_json":"https://pith.science/api/pith-number/ZHAK3FRWS3IADY2L33GNZ3QBIV/graph.json","events_json":"https://pith.science/api/pith-number/ZHAK3FRWS3IADY2L33GNZ3QBIV/events.json","paper":"https://pith.science/paper/ZHAK3FRW"},"agent_actions":{"view_html":"https://pith.science/pith/ZHAK3FRWS3IADY2L33GNZ3QBIV","download_json":"https://pith.science/pith/ZHAK3FRWS3IADY2L33GNZ3QBIV.json","view_paper":"https://pith.science/paper/ZHAK3FRW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1908.11775&json=true","fetch_graph":"https://pith.science/api/pith-number/ZHAK3FRWS3IADY2L33GNZ3QBIV/graph.json","fetch_events":"https://pith.science/api/pith-number/ZHAK3FRWS3IADY2L33GNZ3QBIV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZHAK3FRWS3IADY2L33GNZ3QBIV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZHAK3FRWS3IADY2L33GNZ3QBIV/action/storage_attestation","attest_author":"https://pith.science/pith/ZHAK3FRWS3IADY2L33GNZ3QBIV/action/author_attestation","sign_citation":"https://pith.science/pith/ZHAK3FRWS3IADY2L33GNZ3QBIV/action/citation_signature","submit_replication":"https://pith.science/pith/ZHAK3FRWS3IADY2L33GNZ3QBIV/action/replication_record"}},"created_at":"2026-07-05T00:18:40.014352+00:00","updated_at":"2026-07-05T00:18:40.014352+00:00"}