{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:2LIHOWJIBCM2PZNJGMAGGZAO44","short_pith_number":"pith:2LIHOWJI","schema_version":"1.0","canonical_sha256":"d2d07759280899a7e5a9330063640ee7117606cb9558567988d306dcbde49450","source":{"kind":"arxiv","id":"2210.09221","version":1},"attestation_state":"computed","paper":{"title":"Vision Transformers provably learn spatial structure","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Michael E. Sander, Samy Jelassi, Yuanzhi Li","submitted_at":"2022-10-13T19:53:56Z","abstract_excerpt":"Vision Transformers (ViTs) have achieved comparable or superior performance than Convolutional Neural Networks (CNNs) in computer vision. This empirical breakthrough is even more remarkable since, in contrast to CNNs, ViTs do not embed any visual inductive bias of spatial locality. Yet, recent works have shown that while minimizing their training loss, ViTs specifically learn spatially localized patterns. This raises a central question: how do ViTs learn these patterns by solely minimizing their training loss using gradient-based methods from random initialization? In this paper, we provide so"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2210.09221","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2022-10-13T19:53:56Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"ae76ba0600472b39ff898a90afcd6cd013b19d73202aa939f1e1d84d202845e3","abstract_canon_sha256":"813347e97c9dcf5dfaac3695450ef83f0e48612bc17624713af5b952a00b8fd8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:07:20.250273Z","signature_b64":"rQ44q6mWQrHOthMMz4HnBmDfbmLZ1sJ7lfA6x/gAnGHd+9ilkSbWpDFgTvFcq2TM9jUZ8tRFKldg7J6nQPy5Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d2d07759280899a7e5a9330063640ee7117606cb9558567988d306dcbde49450","last_reissued_at":"2026-07-05T05:07:20.249909Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:07:20.249909Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Vision Transformers provably learn spatial structure","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Michael E. Sander, Samy Jelassi, Yuanzhi Li","submitted_at":"2022-10-13T19:53:56Z","abstract_excerpt":"Vision Transformers (ViTs) have achieved comparable or superior performance than Convolutional Neural Networks (CNNs) in computer vision. This empirical breakthrough is even more remarkable since, in contrast to CNNs, ViTs do not embed any visual inductive bias of spatial locality. Yet, recent works have shown that while minimizing their training loss, ViTs specifically learn spatially localized patterns. This raises a central question: how do ViTs learn these patterns by solely minimizing their training loss using gradient-based methods from random initialization? In this paper, we provide so"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2210.09221","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2210.09221/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2210.09221","created_at":"2026-07-05T05:07:20.249963+00:00"},{"alias_kind":"arxiv_version","alias_value":"2210.09221v1","created_at":"2026-07-05T05:07:20.249963+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2210.09221","created_at":"2026-07-05T05:07:20.249963+00:00"},{"alias_kind":"pith_short_12","alias_value":"2LIHOWJIBCM2","created_at":"2026-07-05T05:07:20.249963+00:00"},{"alias_kind":"pith_short_16","alias_value":"2LIHOWJIBCM2PZNJ","created_at":"2026-07-05T05:07:20.249963+00:00"},{"alias_kind":"pith_short_8","alias_value":"2LIHOWJI","created_at":"2026-07-05T05:07:20.249963+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.06179","citing_title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","ref_index":25,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2LIHOWJIBCM2PZNJGMAGGZAO44","json":"https://pith.science/pith/2LIHOWJIBCM2PZNJGMAGGZAO44.json","graph_json":"https://pith.science/api/pith-number/2LIHOWJIBCM2PZNJGMAGGZAO44/graph.json","events_json":"https://pith.science/api/pith-number/2LIHOWJIBCM2PZNJGMAGGZAO44/events.json","paper":"https://pith.science/paper/2LIHOWJI"},"agent_actions":{"view_html":"https://pith.science/pith/2LIHOWJIBCM2PZNJGMAGGZAO44","download_json":"https://pith.science/pith/2LIHOWJIBCM2PZNJGMAGGZAO44.json","view_paper":"https://pith.science/paper/2LIHOWJI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2210.09221&json=true","fetch_graph":"https://pith.science/api/pith-number/2LIHOWJIBCM2PZNJGMAGGZAO44/graph.json","fetch_events":"https://pith.science/api/pith-number/2LIHOWJIBCM2PZNJGMAGGZAO44/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2LIHOWJIBCM2PZNJGMAGGZAO44/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2LIHOWJIBCM2PZNJGMAGGZAO44/action/storage_attestation","attest_author":"https://pith.science/pith/2LIHOWJIBCM2PZNJGMAGGZAO44/action/author_attestation","sign_citation":"https://pith.science/pith/2LIHOWJIBCM2PZNJGMAGGZAO44/action/citation_signature","submit_replication":"https://pith.science/pith/2LIHOWJIBCM2PZNJGMAGGZAO44/action/replication_record"}},"created_at":"2026-07-05T05:07:20.249963+00:00","updated_at":"2026-07-05T05:07:20.249963+00:00"}