{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:2DZ4ZMWMM7VBVOFB3R3GOA5F7U","short_pith_number":"pith:2DZ4ZMWM","schema_version":"1.0","canonical_sha256":"d0f3ccb2cc67ea1ab8a1dc766703a5fd1254788513b9f4ca378aaa21727fe8aa","source":{"kind":"arxiv","id":"2106.10270","version":2},"attestation_state":"computed","paper":{"title":"How to train your ViT? Data, Augmentation, and Regularization in Vision Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Alexander Kolesnikov, Andreas Steiner, Jakob Uszkoreit, Lucas Beyer, Ross Wightman, Xiaohua Zhai","submitted_at":"2021-06-18T17:58:20Z","abstract_excerpt":"Vision Transformers (ViT) have been shown to attain highly competitive performance for a wide range of vision applications, such as image classification, object detection and semantic image segmentation. In comparison to convolutional neural networks, the Vision Transformer's weaker inductive bias is generally found to cause an increased reliance on model regularization or data augmentation (\"AugReg\" for short) when training on smaller training datasets. We conduct a systematic empirical study in order to better understand the interplay between the amount of training data, AugReg, model size a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2106.10270","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2021-06-18T17:58:20Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"987ffbd054f777def434f35be409a1d07e82148eb51dc9a7eba25742b81b9d6d","abstract_canon_sha256":"b981b0fbe5e791b16e829a1f9871d9cd2a3b72e7606d4f29032c66354a4529ea"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:34:12.714333Z","signature_b64":"SrmIJQuNzEeNP7i3SaGg3GGAPMIu76BDESJIe3QQHl7oLLKlxMMSuaFEgK4zUKqAwik1IpBzvqP+tcs6g6TkBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d0f3ccb2cc67ea1ab8a1dc766703a5fd1254788513b9f4ca378aaa21727fe8aa","last_reissued_at":"2026-07-05T04:34:12.713835Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:34:12.713835Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"How to train your ViT? Data, Augmentation, and Regularization in Vision Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Alexander Kolesnikov, Andreas Steiner, Jakob Uszkoreit, Lucas Beyer, Ross Wightman, Xiaohua Zhai","submitted_at":"2021-06-18T17:58:20Z","abstract_excerpt":"Vision Transformers (ViT) have been shown to attain highly competitive performance for a wide range of vision applications, such as image classification, object detection and semantic image segmentation. In comparison to convolutional neural networks, the Vision Transformer's weaker inductive bias is generally found to cause an increased reliance on model regularization or data augmentation (\"AugReg\" for short) when training on smaller training datasets. We conduct a systematic empirical study in order to better understand the interplay between the amount of training data, AugReg, model size a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2106.10270","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2106.10270/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2106.10270","created_at":"2026-07-05T04:34:12.713904+00:00"},{"alias_kind":"arxiv_version","alias_value":"2106.10270v2","created_at":"2026-07-05T04:34:12.713904+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2106.10270","created_at":"2026-07-05T04:34:12.713904+00:00"},{"alias_kind":"pith_short_12","alias_value":"2DZ4ZMWMM7VB","created_at":"2026-07-05T04:34:12.713904+00:00"},{"alias_kind":"pith_short_16","alias_value":"2DZ4ZMWMM7VBVOFB","created_at":"2026-07-05T04:34:12.713904+00:00"},{"alias_kind":"pith_short_8","alias_value":"2DZ4ZMWM","created_at":"2026-07-05T04:34:12.713904+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.27527","citing_title":"Large Language Model Teaches Visual Students: Cross-Modality Transfer of Fine-Grained Conceptual Knowledge","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23719","citing_title":"Weierstrass Positional Encoding for Vision Transformers","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22372","citing_title":"ASAP: Attention Sink Anchored Pruning","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2603.13652","citing_title":"Causal Attribution via Activation Patching","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2303.15343","citing_title":"Sigmoid Loss for Language Image Pre-Training","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2309.16671","citing_title":"Demystifying CLIP Data","ref_index":80,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14521","citing_title":"Enjoy Your Layer Normalization with the Computational Efficiency of RMSNorm","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2403.14608","citing_title":"Parameter-Efficient Fine-Tuning for Large Models: A Comprehensive Survey","ref_index":185,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18094","citing_title":"Decision-Aware Attention Propagation for Vision Transformer Explainability","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2410.24164","citing_title":"$\\pi_0$: A Vision-Language-Action Flow Model for General Robot Control","ref_index":47,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2DZ4ZMWMM7VBVOFB3R3GOA5F7U","json":"https://pith.science/pith/2DZ4ZMWMM7VBVOFB3R3GOA5F7U.json","graph_json":"https://pith.science/api/pith-number/2DZ4ZMWMM7VBVOFB3R3GOA5F7U/graph.json","events_json":"https://pith.science/api/pith-number/2DZ4ZMWMM7VBVOFB3R3GOA5F7U/events.json","paper":"https://pith.science/paper/2DZ4ZMWM"},"agent_actions":{"view_html":"https://pith.science/pith/2DZ4ZMWMM7VBVOFB3R3GOA5F7U","download_json":"https://pith.science/pith/2DZ4ZMWMM7VBVOFB3R3GOA5F7U.json","view_paper":"https://pith.science/paper/2DZ4ZMWM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2106.10270&json=true","fetch_graph":"https://pith.science/api/pith-number/2DZ4ZMWMM7VBVOFB3R3GOA5F7U/graph.json","fetch_events":"https://pith.science/api/pith-number/2DZ4ZMWMM7VBVOFB3R3GOA5F7U/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2DZ4ZMWMM7VBVOFB3R3GOA5F7U/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2DZ4ZMWMM7VBVOFB3R3GOA5F7U/action/storage_attestation","attest_author":"https://pith.science/pith/2DZ4ZMWMM7VBVOFB3R3GOA5F7U/action/author_attestation","sign_citation":"https://pith.science/pith/2DZ4ZMWMM7VBVOFB3R3GOA5F7U/action/citation_signature","submit_replication":"https://pith.science/pith/2DZ4ZMWMM7VBVOFB3R3GOA5F7U/action/replication_record"}},"created_at":"2026-07-05T04:34:12.713904+00:00","updated_at":"2026-07-05T04:34:12.713904+00:00"}