{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:RWUSI7K66XTQ4NXB4NR2HUP6AI","short_pith_number":"pith:RWUSI7K6","schema_version":"1.0","canonical_sha256":"8da9247d5ef5e70e36e1e363a3d1fe0211ef707d6b9c8bcae644c1ac835310ba","source":{"kind":"arxiv","id":"2202.06709","version":4},"attestation_state":"computed","paper":{"title":"How Do Vision Transformers Work?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Namuk Park, Songkuk Kim","submitted_at":"2022-02-14T13:58:43Z","abstract_excerpt":"The success of multi-head self-attentions (MSAs) for computer vision is now indisputable. However, little is known about how MSAs work. We present fundamental explanations to help better understand the nature of MSAs. In particular, we demonstrate the following properties of MSAs and Vision Transformers (ViTs): (1) MSAs improve not only accuracy but also generalization by flattening the loss landscapes. Such improvement is primarily attributable to their data specificity, not long-range dependency. On the other hand, ViTs suffer from non-convex losses. Large datasets and loss landscape smoothi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2202.06709","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2022-02-14T13:58:43Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"fb08097970f9f16fa1be538e7c26e2109fdee1f2b02cceb5e8a2e7627e376a4a","abstract_canon_sha256":"54236f66b7efb51852fbe655ab53d0371d4bf9c141b96e005777519230836b0c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:30:07.327202Z","signature_b64":"6hqTLmO++CNX8sARPkK2lCNgomGZsu+Ulb3CHneZ4KNZTTulXZ93CW+KetjnFaQ6n8sn+2aNAX8Vtn8JHQEBDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8da9247d5ef5e70e36e1e363a3d1fe0211ef707d6b9c8bcae644c1ac835310ba","last_reissued_at":"2026-07-05T04:30:07.326637Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:30:07.326637Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"How Do Vision Transformers Work?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Namuk Park, Songkuk Kim","submitted_at":"2022-02-14T13:58:43Z","abstract_excerpt":"The success of multi-head self-attentions (MSAs) for computer vision is now indisputable. However, little is known about how MSAs work. We present fundamental explanations to help better understand the nature of MSAs. In particular, we demonstrate the following properties of MSAs and Vision Transformers (ViTs): (1) MSAs improve not only accuracy but also generalization by flattening the loss landscapes. Such improvement is primarily attributable to their data specificity, not long-range dependency. On the other hand, ViTs suffer from non-convex losses. Large datasets and loss landscape smoothi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2202.06709","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2202.06709/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2202.06709","created_at":"2026-07-05T04:30:07.326702+00:00"},{"alias_kind":"arxiv_version","alias_value":"2202.06709v4","created_at":"2026-07-05T04:30:07.326702+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2202.06709","created_at":"2026-07-05T04:30:07.326702+00:00"},{"alias_kind":"pith_short_12","alias_value":"RWUSI7K66XTQ","created_at":"2026-07-05T04:30:07.326702+00:00"},{"alias_kind":"pith_short_16","alias_value":"RWUSI7K66XTQ4NXB","created_at":"2026-07-05T04:30:07.326702+00:00"},{"alias_kind":"pith_short_8","alias_value":"RWUSI7K6","created_at":"2026-07-05T04:30:07.326702+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":16,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05176","citing_title":"FSDC-DETR: A Frequency-Spatial Domain Collaborative DETR for Small Object Detection","ref_index":41,"is_internal_anchor":true},{"citing_arxiv_id":"2606.21072","citing_title":"SqLinear: Balanced Square Partitioning Makes Linear Interaction Sufficient for Large-Scale Traffic Forecasting","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20077","citing_title":"The Hidden Evolution of Disguised Visual Context inside the VLM","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20044","citing_title":"FUSE: Frequency-domain Unification and Spectral Energy Alignment for Multi-modal Object Re-Identification","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01906","citing_title":"SFKD: Spatial--Frequency Joint-Aware Heterogeneous Knowledge Distillation via Multi-Level Wavelet Spectral Interaction","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00223","citing_title":"Does Your ViT Still Need U-Net for Segmentation?","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00620","citing_title":"Identifying Latent Concepts and Structures for Generalized Category Discovery","ref_index":213,"is_internal_anchor":false},{"citing_arxiv_id":"2504.16455","citing_title":"Cross Paradigm Representation and Alignment Transformer for Image Deraining","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18328","citing_title":"CineMatte: Background Matting for Virtual Production and Beyond","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08935","citing_title":"PnP-Corrector: A Universal Correction Framework for Coupled Spatiotemporal Forecasting","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12491","citing_title":"Elastic Attention Cores for Scalable Vision Transformers","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08935","citing_title":"PnP-Corrector: A Universal Correction Framework for Coupled Spatiotemporal Forecasting","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09905","citing_title":"Rethinking Random Transformers as Adaptive Sequence Smoothers for Sleep Staging","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06783","citing_title":"Insights from Visual Cognition: Understanding Human Action Dynamics with Overall Glance and Refined Gaze Transformer","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14526","citing_title":"FreqTrack: Frequency Learning based Vision Transformer for RGB-Event Object Tracking","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00510","citing_title":"Scale-Aware Adversarial Analysis: A Diagnostic for Generative AI in Multiscale Complex Systems","ref_index":112,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RWUSI7K66XTQ4NXB4NR2HUP6AI","json":"https://pith.science/pith/RWUSI7K66XTQ4NXB4NR2HUP6AI.json","graph_json":"https://pith.science/api/pith-number/RWUSI7K66XTQ4NXB4NR2HUP6AI/graph.json","events_json":"https://pith.science/api/pith-number/RWUSI7K66XTQ4NXB4NR2HUP6AI/events.json","paper":"https://pith.science/paper/RWUSI7K6"},"agent_actions":{"view_html":"https://pith.science/pith/RWUSI7K66XTQ4NXB4NR2HUP6AI","download_json":"https://pith.science/pith/RWUSI7K66XTQ4NXB4NR2HUP6AI.json","view_paper":"https://pith.science/paper/RWUSI7K6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2202.06709&json=true","fetch_graph":"https://pith.science/api/pith-number/RWUSI7K66XTQ4NXB4NR2HUP6AI/graph.json","fetch_events":"https://pith.science/api/pith-number/RWUSI7K66XTQ4NXB4NR2HUP6AI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RWUSI7K66XTQ4NXB4NR2HUP6AI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RWUSI7K66XTQ4NXB4NR2HUP6AI/action/storage_attestation","attest_author":"https://pith.science/pith/RWUSI7K66XTQ4NXB4NR2HUP6AI/action/author_attestation","sign_citation":"https://pith.science/pith/RWUSI7K66XTQ4NXB4NR2HUP6AI/action/citation_signature","submit_replication":"https://pith.science/pith/RWUSI7K66XTQ4NXB4NR2HUP6AI/action/replication_record"}},"created_at":"2026-07-05T04:30:07.326702+00:00","updated_at":"2026-07-05T04:30:07.326702+00:00"}