{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:USILYVDEKGZGKVYV5PUQIIX4ZC","short_pith_number":"pith:USILYVDE","schema_version":"1.0","canonical_sha256":"a490bc546451b2655715ebe90422fcc8aca9b260d40e858f248a6e1b5b9babff","source":{"kind":"arxiv","id":"2203.05962","version":1},"attestation_state":"computed","paper":{"title":"Anti-Oversmoothing in Deep Vision Transformers via the Fourier Domain Analysis: From Theory to Practice","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Peihao Wang, Tianlong Chen, Wenqing Zheng, Zhangyang Wang","submitted_at":"2022-03-09T23:55:24Z","abstract_excerpt":"Vision Transformer (ViT) has recently demonstrated promise in computer vision problems. However, unlike Convolutional Neural Networks (CNN), it is known that the performance of ViT saturates quickly with depth increasing, due to the observed attention collapse or patch uniformity. Despite a couple of empirical solutions, a rigorous framework studying on this scalability issue remains elusive. In this paper, we first establish a rigorous theory framework to analyze ViT features from the Fourier spectrum domain. We show that the self-attention mechanism inherently amounts to a low-pass filter, w"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2203.05962","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2022-03-09T23:55:24Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"c8392d0bf827324e83d8c272b1b2cdda215c74481301ec181511641b2e28eb17","abstract_canon_sha256":"67ab2d669bc1b825da76f2888a1104212bdb32e790952288dc387968acfaa741"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:04:11.377757Z","signature_b64":"3qqn5A2PAtUoszqy2TaKmv8XrADYtq0eYaoJfEGc33HKIkwD52OdGe2UCGNigMjtbDpxiW5ndK1nPkti/FAdAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a490bc546451b2655715ebe90422fcc8aca9b260d40e858f248a6e1b5b9babff","last_reissued_at":"2026-07-05T04:04:11.377276Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:04:11.377276Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Anti-Oversmoothing in Deep Vision Transformers via the Fourier Domain Analysis: From Theory to Practice","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Peihao Wang, Tianlong Chen, Wenqing Zheng, Zhangyang Wang","submitted_at":"2022-03-09T23:55:24Z","abstract_excerpt":"Vision Transformer (ViT) has recently demonstrated promise in computer vision problems. However, unlike Convolutional Neural Networks (CNN), it is known that the performance of ViT saturates quickly with depth increasing, due to the observed attention collapse or patch uniformity. Despite a couple of empirical solutions, a rigorous framework studying on this scalability issue remains elusive. In this paper, we first establish a rigorous theory framework to analyze ViT features from the Fourier spectrum domain. We show that the self-attention mechanism inherently amounts to a low-pass filter, w"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2203.05962","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2203.05962/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2203.05962","created_at":"2026-07-05T04:04:11.377340+00:00"},{"alias_kind":"arxiv_version","alias_value":"2203.05962v1","created_at":"2026-07-05T04:04:11.377340+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2203.05962","created_at":"2026-07-05T04:04:11.377340+00:00"},{"alias_kind":"pith_short_12","alias_value":"USILYVDEKGZG","created_at":"2026-07-05T04:04:11.377340+00:00"},{"alias_kind":"pith_short_16","alias_value":"USILYVDEKGZGKVYV","created_at":"2026-07-05T04:04:11.377340+00:00"},{"alias_kind":"pith_short_8","alias_value":"USILYVDE","created_at":"2026-07-05T04:04:11.377340+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05176","citing_title":"FSDC-DETR: A Frequency-Spatial Domain Collaborative DETR for Small Object Detection","ref_index":69,"is_internal_anchor":true},{"citing_arxiv_id":"2606.03795","citing_title":"Beyond Compression: Quantifying Spectral Accessibility in Vision Representations","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2603.05117","citing_title":"SeedPolicy: Horizon Scaling via Self-Evolving Diffusion Policy for Robot Manipulation","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09905","citing_title":"Rethinking Random Transformers as Adaptive Sequence Smoothers for Sleep Staging","ref_index":84,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/USILYVDEKGZGKVYV5PUQIIX4ZC","json":"https://pith.science/pith/USILYVDEKGZGKVYV5PUQIIX4ZC.json","graph_json":"https://pith.science/api/pith-number/USILYVDEKGZGKVYV5PUQIIX4ZC/graph.json","events_json":"https://pith.science/api/pith-number/USILYVDEKGZGKVYV5PUQIIX4ZC/events.json","paper":"https://pith.science/paper/USILYVDE"},"agent_actions":{"view_html":"https://pith.science/pith/USILYVDEKGZGKVYV5PUQIIX4ZC","download_json":"https://pith.science/pith/USILYVDEKGZGKVYV5PUQIIX4ZC.json","view_paper":"https://pith.science/paper/USILYVDE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2203.05962&json=true","fetch_graph":"https://pith.science/api/pith-number/USILYVDEKGZGKVYV5PUQIIX4ZC/graph.json","fetch_events":"https://pith.science/api/pith-number/USILYVDEKGZGKVYV5PUQIIX4ZC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/USILYVDEKGZGKVYV5PUQIIX4ZC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/USILYVDEKGZGKVYV5PUQIIX4ZC/action/storage_attestation","attest_author":"https://pith.science/pith/USILYVDEKGZGKVYV5PUQIIX4ZC/action/author_attestation","sign_citation":"https://pith.science/pith/USILYVDEKGZGKVYV5PUQIIX4ZC/action/citation_signature","submit_replication":"https://pith.science/pith/USILYVDEKGZGKVYV5PUQIIX4ZC/action/replication_record"}},"created_at":"2026-07-05T04:04:11.377340+00:00","updated_at":"2026-07-05T04:04:11.377340+00:00"}