{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:F7URNVIWAN3DJBOJ7XN34U7CP7","short_pith_number":"pith:F7URNVIW","schema_version":"1.0","canonical_sha256":"2fe916d51603763485c9fddbbe53e27fd9addb58c8d5558ff556004638d4797f","source":{"kind":"arxiv","id":"2104.02057","version":4},"attestation_state":"computed","paper":{"title":"An Empirical Study of Training Self-Supervised Vision Transformers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Kaiming He, Saining Xie, Xinlei Chen","submitted_at":"2021-04-05T17:59:40Z","abstract_excerpt":"This paper does not describe a novel method. Instead, it studies a straightforward, incremental, yet must-know baseline given the recent progress in computer vision: self-supervised learning for Vision Transformers (ViT). While the training recipes for standard convolutional networks have been highly mature and robust, the recipes for ViT are yet to be built, especially in the self-supervised scenarios where training becomes more challenging. In this work, we go back to basics and investigate the effects of several fundamental components for training self-supervised ViT. We observe that instab"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2104.02057","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2021-04-05T17:59:40Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"7667c8373ac414bff9e24bac1fe8c29f2d27e99a3b9865a6b744d5597d1cd014","abstract_canon_sha256":"b8a623b958db4f0c410fb151c3e3146c6b32c6e8bed249af498aa70602091f85"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:05:48.901749Z","signature_b64":"2sXpRdA0yL0I1sKS7X3GZShV5bQf11RoWatumLWr6BQ9vZN2yycuKIPcolRRcT1Tvw71ITITBVXR4vqKol7dBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2fe916d51603763485c9fddbbe53e27fd9addb58c8d5558ff556004638d4797f","last_reissued_at":"2026-07-05T03:05:48.901222Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:05:48.901222Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"An Empirical Study of Training Self-Supervised Vision Transformers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Kaiming He, Saining Xie, Xinlei Chen","submitted_at":"2021-04-05T17:59:40Z","abstract_excerpt":"This paper does not describe a novel method. Instead, it studies a straightforward, incremental, yet must-know baseline given the recent progress in computer vision: self-supervised learning for Vision Transformers (ViT). While the training recipes for standard convolutional networks have been highly mature and robust, the recipes for ViT are yet to be built, especially in the self-supervised scenarios where training becomes more challenging. In this work, we go back to basics and investigate the effects of several fundamental components for training self-supervised ViT. We observe that instab"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2104.02057","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2104.02057/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2104.02057","created_at":"2026-07-05T03:05:48.901293+00:00"},{"alias_kind":"arxiv_version","alias_value":"2104.02057v4","created_at":"2026-07-05T03:05:48.901293+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2104.02057","created_at":"2026-07-05T03:05:48.901293+00:00"},{"alias_kind":"pith_short_12","alias_value":"F7URNVIWAN3D","created_at":"2026-07-05T03:05:48.901293+00:00"},{"alias_kind":"pith_short_16","alias_value":"F7URNVIWAN3DJBOJ","created_at":"2026-07-05T03:05:48.901293+00:00"},{"alias_kind":"pith_short_8","alias_value":"F7URNVIW","created_at":"2026-07-05T03:05:48.901293+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2503.16683","citing_title":"GAIR: Location-Aware Self-Supervised Contrastive Pre-Training with Geo-Aligned Implicit Representations","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18878","citing_title":"Prognostic Value of Lung Ultrasound Biomarkers for Readmission Risk in Congestive Heart Failure: A Pilot Data-Driven Analysis","ref_index":111,"is_internal_anchor":false},{"citing_arxiv_id":"2508.09691","citing_title":"PaCo-FR: Patch-Pixel Aligned End-to-End Codebook Learning for Facial Representation Pre-training","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2309.16797","citing_title":"Promptbreeder: Self-Referential Self-Improvement Via Prompt Evolution","ref_index":74,"is_internal_anchor":false},{"citing_arxiv_id":"2312.17090","citing_title":"Q-Align: Teaching LMMs for Visual Scoring via Discrete Text-Defined Levels","ref_index":129,"is_internal_anchor":false},{"citing_arxiv_id":"2106.08254","citing_title":"BEiT: BERT Pre-Training of Image Transformers","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2404.08471","citing_title":"Revisiting Feature Prediction for Learning Visual Representations from Video","ref_index":230,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03245","citing_title":"Text-Conditional JEPA for Learning Semantically Rich Visual Representations","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18251","citing_title":"Style-Based Neural Architectures for Real-Time Weather Classification","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13947","citing_title":"Heuristic Style Transfer for Real-Time, Efficient Weather Attribute Detection","ref_index":58,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/F7URNVIWAN3DJBOJ7XN34U7CP7","json":"https://pith.science/pith/F7URNVIWAN3DJBOJ7XN34U7CP7.json","graph_json":"https://pith.science/api/pith-number/F7URNVIWAN3DJBOJ7XN34U7CP7/graph.json","events_json":"https://pith.science/api/pith-number/F7URNVIWAN3DJBOJ7XN34U7CP7/events.json","paper":"https://pith.science/paper/F7URNVIW"},"agent_actions":{"view_html":"https://pith.science/pith/F7URNVIWAN3DJBOJ7XN34U7CP7","download_json":"https://pith.science/pith/F7URNVIWAN3DJBOJ7XN34U7CP7.json","view_paper":"https://pith.science/paper/F7URNVIW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2104.02057&json=true","fetch_graph":"https://pith.science/api/pith-number/F7URNVIWAN3DJBOJ7XN34U7CP7/graph.json","fetch_events":"https://pith.science/api/pith-number/F7URNVIWAN3DJBOJ7XN34U7CP7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/F7URNVIWAN3DJBOJ7XN34U7CP7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/F7URNVIWAN3DJBOJ7XN34U7CP7/action/storage_attestation","attest_author":"https://pith.science/pith/F7URNVIWAN3DJBOJ7XN34U7CP7/action/author_attestation","sign_citation":"https://pith.science/pith/F7URNVIWAN3DJBOJ7XN34U7CP7/action/citation_signature","submit_replication":"https://pith.science/pith/F7URNVIWAN3DJBOJ7XN34U7CP7/action/replication_record"}},"created_at":"2026-07-05T03:05:48.901293+00:00","updated_at":"2026-07-05T03:05:48.901293+00:00"}