{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:HTZQOELVOSNJBDPC5OTKYED5DK","short_pith_number":"pith:HTZQOELV","schema_version":"1.0","canonical_sha256":"3cf3071175749a908de2eba6ac107d1a80a68ac0d465da493a115a297a53bf9f","source":{"kind":"arxiv","id":"2101.11986","version":3},"attestation_state":"computed","paper":{"title":"Tokens-to-Token ViT: Training Vision Transformers from Scratch on ImageNet","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Francis EH Tay, Jiashi Feng, Li Yuan, Shuicheng Yan, Tao Wang, Weihao Yu, Yujun Shi, Yunpeng Chen, Zihang Jiang","submitted_at":"2021-01-28T13:25:28Z","abstract_excerpt":"Transformers, which are popular for language modeling, have been explored for solving vision tasks recently, e.g., the Vision Transformer (ViT) for image classification. The ViT model splits each image into a sequence of tokens with fixed length and then applies multiple Transformer layers to model their global relation for classification. However, ViT achieves inferior performance to CNNs when trained from scratch on a midsize dataset like ImageNet. We find it is because: 1) the simple tokenization of input images fails to model the important local structure such as edges and lines among neig"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2101.11986","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2021-01-28T13:25:28Z","cross_cats_sorted":[],"title_canon_sha256":"55c68d805b902c41e21c3892a3e2758c24c55ef8746a44ccbf7ca44a79c7c4bf","abstract_canon_sha256":"87c9e82a18927b12d398c3b1566aa012714b7b67e86d4d3ace1927944b98f786"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:36:01.142613Z","signature_b64":"H5DnWY1b84TnR2zNVp4yWHPxXFbz9/tJBWKL9H5ZqdNBNX+B1MVEy5u7+k6TvqvXJH9pxo6ZoihQ89vZVEvUDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3cf3071175749a908de2eba6ac107d1a80a68ac0d465da493a115a297a53bf9f","last_reissued_at":"2026-07-05T03:36:01.142123Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:36:01.142123Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Tokens-to-Token ViT: Training Vision Transformers from Scratch on ImageNet","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Francis EH Tay, Jiashi Feng, Li Yuan, Shuicheng Yan, Tao Wang, Weihao Yu, Yujun Shi, Yunpeng Chen, Zihang Jiang","submitted_at":"2021-01-28T13:25:28Z","abstract_excerpt":"Transformers, which are popular for language modeling, have been explored for solving vision tasks recently, e.g., the Vision Transformer (ViT) for image classification. The ViT model splits each image into a sequence of tokens with fixed length and then applies multiple Transformer layers to model their global relation for classification. However, ViT achieves inferior performance to CNNs when trained from scratch on a midsize dataset like ImageNet. We find it is because: 1) the simple tokenization of input images fails to model the important local structure such as edges and lines among neig"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2101.11986","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2101.11986/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2101.11986","created_at":"2026-07-05T03:36:01.142177+00:00"},{"alias_kind":"arxiv_version","alias_value":"2101.11986v3","created_at":"2026-07-05T03:36:01.142177+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2101.11986","created_at":"2026-07-05T03:36:01.142177+00:00"},{"alias_kind":"pith_short_12","alias_value":"HTZQOELVOSNJ","created_at":"2026-07-05T03:36:01.142177+00:00"},{"alias_kind":"pith_short_16","alias_value":"HTZQOELVOSNJBDPC","created_at":"2026-07-05T03:36:01.142177+00:00"},{"alias_kind":"pith_short_8","alias_value":"HTZQOELV","created_at":"2026-07-05T03:36:01.142177+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.20732","citing_title":"Deep Attention Reweighting: Post-Hoc Attention-Based Feature Aggregation in CNNs for Disentangling Core and Spurious Features under Spurious Correlations","ref_index":78,"is_internal_anchor":false},{"citing_arxiv_id":"2103.14030","citing_title":"Swin Transformer: Hierarchical Vision Transformer using Shifted Windows","ref_index":72,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HTZQOELVOSNJBDPC5OTKYED5DK","json":"https://pith.science/pith/HTZQOELVOSNJBDPC5OTKYED5DK.json","graph_json":"https://pith.science/api/pith-number/HTZQOELVOSNJBDPC5OTKYED5DK/graph.json","events_json":"https://pith.science/api/pith-number/HTZQOELVOSNJBDPC5OTKYED5DK/events.json","paper":"https://pith.science/paper/HTZQOELV"},"agent_actions":{"view_html":"https://pith.science/pith/HTZQOELVOSNJBDPC5OTKYED5DK","download_json":"https://pith.science/pith/HTZQOELVOSNJBDPC5OTKYED5DK.json","view_paper":"https://pith.science/paper/HTZQOELV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2101.11986&json=true","fetch_graph":"https://pith.science/api/pith-number/HTZQOELVOSNJBDPC5OTKYED5DK/graph.json","fetch_events":"https://pith.science/api/pith-number/HTZQOELVOSNJBDPC5OTKYED5DK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HTZQOELVOSNJBDPC5OTKYED5DK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HTZQOELVOSNJBDPC5OTKYED5DK/action/storage_attestation","attest_author":"https://pith.science/pith/HTZQOELVOSNJBDPC5OTKYED5DK/action/author_attestation","sign_citation":"https://pith.science/pith/HTZQOELVOSNJBDPC5OTKYED5DK/action/citation_signature","submit_replication":"https://pith.science/pith/HTZQOELVOSNJBDPC5OTKYED5DK/action/replication_record"}},"created_at":"2026-07-05T03:36:01.142177+00:00","updated_at":"2026-07-05T03:36:01.142177+00:00"}