{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:QGPWGSETNZKCQXHHC5J7T4N2UW","short_pith_number":"pith:QGPWGSET","schema_version":"1.0","canonical_sha256":"819f6348936e54285ce71753f9f1baa5b96b61769eec069db3d8c8c6ab03040f","source":{"kind":"arxiv","id":"2106.04560","version":2},"attestation_state":"computed","paper":{"title":"Scaling Vision Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Alexander Kolesnikov, Lucas Beyer, Neil Houlsby, Xiaohua Zhai","submitted_at":"2021-06-08T17:47:39Z","abstract_excerpt":"Attention-based neural networks such as the Vision Transformer (ViT) have recently attained state-of-the-art results on many computer vision benchmarks. Scale is a primary ingredient in attaining excellent results, therefore, understanding a model's scaling properties is a key to designing future generations effectively. While the laws for scaling Transformer language models have been studied, it is unknown how Vision Transformers scale. To address this, we scale ViT models and data, both up and down, and characterize the relationships between error rate, data, and compute. Along the way, we r"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2106.04560","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2021-06-08T17:47:39Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"ab69646605417351ea5aef38b6b1177874142266e7c54dc3ab80bc81bea84e9d","abstract_canon_sha256":"e44abe2f8e98a90449aee672ec46a4404e358ede463246862df550951932b5d8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:32:49.824932Z","signature_b64":"olu46bMJufQajnfn1RkhkfQ0jIEV8SpgSgATGudrGft+mNL1cBCXLXi9jVt8m35p+6suslNlIFUeAlEEvvCgCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"819f6348936e54285ce71753f9f1baa5b96b61769eec069db3d8c8c6ab03040f","last_reissued_at":"2026-07-05T04:32:49.824438Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:32:49.824438Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scaling Vision Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Alexander Kolesnikov, Lucas Beyer, Neil Houlsby, Xiaohua Zhai","submitted_at":"2021-06-08T17:47:39Z","abstract_excerpt":"Attention-based neural networks such as the Vision Transformer (ViT) have recently attained state-of-the-art results on many computer vision benchmarks. Scale is a primary ingredient in attaining excellent results, therefore, understanding a model's scaling properties is a key to designing future generations effectively. While the laws for scaling Transformer language models have been studied, it is unknown how Vision Transformers scale. To address this, we scale ViT models and data, both up and down, and characterize the relationships between error rate, data, and compute. Along the way, we r"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2106.04560","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2106.04560/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2106.04560","created_at":"2026-07-05T04:32:49.824500+00:00"},{"alias_kind":"arxiv_version","alias_value":"2106.04560v2","created_at":"2026-07-05T04:32:49.824500+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2106.04560","created_at":"2026-07-05T04:32:49.824500+00:00"},{"alias_kind":"pith_short_12","alias_value":"QGPWGSETNZKC","created_at":"2026-07-05T04:32:49.824500+00:00"},{"alias_kind":"pith_short_16","alias_value":"QGPWGSETNZKCQXHH","created_at":"2026-07-05T04:32:49.824500+00:00"},{"alias_kind":"pith_short_8","alias_value":"QGPWGSET","created_at":"2026-07-05T04:32:49.824500+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":19,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25971","citing_title":"Improving Neural Network Training by Decoupling the Magnitude and Direction of Weight Vectors","ref_index":127,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02794","citing_title":"Scaling Laws for Neural-Network Quantum States","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26248","citing_title":"Unified Neural Scaling Laws","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23189","citing_title":"Empirical Bayes Conformal Prediction for Vision and Language Models","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18058","citing_title":"Threats to Arabic Handwriting Recognition: Investigating Black-Box Adversarial Attacks on embedded ConvNet models","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2408.00724","citing_title":"Inference Scaling Laws: An Empirical Analysis of Compute-Optimal Inference for Problem-Solving with Language Models","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2303.09540","citing_title":"SemDeDup: Data-efficient learning at web-scale through semantic deduplication","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2511.16719","citing_title":"SAM 3: Segment Anything with Concepts","ref_index":156,"is_internal_anchor":false},{"citing_arxiv_id":"2601.10791","citing_title":"OmniMol: Transferring Particle Physics Knowledge to Molecular Dynamics with Point-Edge Transformers","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2111.11432","citing_title":"Florence: A New Foundation Model for Computer Vision","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2309.16671","citing_title":"Demystifying CLIP Data","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2502.02737","citing_title":"SmolLM2: When Smol Goes Big -- Data-Centric Training of a Small Language Model","ref_index":251,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03522","citing_title":"Physics-Informed Transformer for Real-Time High-Fidelity Topology Optimization","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2210.08402","citing_title":"LAION-5B: An open large-scale dataset for training next generation image-text models","ref_index":97,"is_internal_anchor":false},{"citing_arxiv_id":"2210.08402","citing_title":"LAION-5B: An open large-scale dataset for training next generation image-text models","ref_index":96,"is_internal_anchor":false},{"citing_arxiv_id":"2106.08254","citing_title":"BEiT: BERT Pre-Training of Image Transformers","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2111.02114","citing_title":"LAION-400M: Open Dataset of CLIP-Filtered 400 Million Image-Text Pairs","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2204.14198","citing_title":"Flamingo: a Visual Language Model for Few-Shot Learning","ref_index":146,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10021","citing_title":"Masked Contrastive Pre-Training Improves Music Audio Key Detection","ref_index":25,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QGPWGSETNZKCQXHHC5J7T4N2UW","json":"https://pith.science/pith/QGPWGSETNZKCQXHHC5J7T4N2UW.json","graph_json":"https://pith.science/api/pith-number/QGPWGSETNZKCQXHHC5J7T4N2UW/graph.json","events_json":"https://pith.science/api/pith-number/QGPWGSETNZKCQXHHC5J7T4N2UW/events.json","paper":"https://pith.science/paper/QGPWGSET"},"agent_actions":{"view_html":"https://pith.science/pith/QGPWGSETNZKCQXHHC5J7T4N2UW","download_json":"https://pith.science/pith/QGPWGSETNZKCQXHHC5J7T4N2UW.json","view_paper":"https://pith.science/paper/QGPWGSET","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2106.04560&json=true","fetch_graph":"https://pith.science/api/pith-number/QGPWGSETNZKCQXHHC5J7T4N2UW/graph.json","fetch_events":"https://pith.science/api/pith-number/QGPWGSETNZKCQXHHC5J7T4N2UW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QGPWGSETNZKCQXHHC5J7T4N2UW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QGPWGSETNZKCQXHHC5J7T4N2UW/action/storage_attestation","attest_author":"https://pith.science/pith/QGPWGSETNZKCQXHHC5J7T4N2UW/action/author_attestation","sign_citation":"https://pith.science/pith/QGPWGSETNZKCQXHHC5J7T4N2UW/action/citation_signature","submit_replication":"https://pith.science/pith/QGPWGSETNZKCQXHHC5J7T4N2UW/action/replication_record"}},"created_at":"2026-07-05T04:32:49.824500+00:00","updated_at":"2026-07-05T04:32:49.824500+00:00"}