{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:XIUH7MZC4Q4WJMGTYYJ2AH4QRE","short_pith_number":"pith:XIUH7MZC","schema_version":"1.0","canonical_sha256":"ba287fb322e43964b0d3c613a01f908926359a657be37a7c719619436edaf81a","source":{"kind":"arxiv","id":"2106.14881","version":3},"attestation_state":"computed","paper":{"title":"Early Convolutions Help Transformers See Better","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Eric Mintun, Mannat Singh, Piotr Doll\\'ar, Ross Girshick, Tete Xiao, Trevor Darrell","submitted_at":"2021-06-28T17:59:33Z","abstract_excerpt":"Vision transformer (ViT) models exhibit substandard optimizability. In particular, they are sensitive to the choice of optimizer (AdamW vs. SGD), optimizer hyperparameters, and training schedule length. In comparison, modern convolutional neural networks are easier to optimize. Why is this the case? In this work, we conjecture that the issue lies with the patchify stem of ViT models, which is implemented by a stride-p p*p convolution (p=16 by default) applied to the input image. This large-kernel plus large-stride convolution runs counter to typical design choices of convolutional layers in ne"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2106.14881","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2021-06-28T17:59:33Z","cross_cats_sorted":[],"title_canon_sha256":"96dcd19035b316638979c7a18193c6ab4c5b0a9dbb5fa6b789f962ac2795b44a","abstract_canon_sha256":"19834498e30aafa6442bb3c3e5d69ef58b44b5383ad7cf3bc504f004e1718e40"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:25:32.497340Z","signature_b64":"01v487iPDF/r3iNBWuFK5O5uaScTEGzBPb/pTGLrLvEzLXCXjiK1hG0TJqlEUDUtIogfnny58PN8COMyjFbeBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ba287fb322e43964b0d3c613a01f908926359a657be37a7c719619436edaf81a","last_reissued_at":"2026-07-05T03:25:32.496944Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:25:32.496944Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Early Convolutions Help Transformers See Better","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Eric Mintun, Mannat Singh, Piotr Doll\\'ar, Ross Girshick, Tete Xiao, Trevor Darrell","submitted_at":"2021-06-28T17:59:33Z","abstract_excerpt":"Vision transformer (ViT) models exhibit substandard optimizability. In particular, they are sensitive to the choice of optimizer (AdamW vs. SGD), optimizer hyperparameters, and training schedule length. In comparison, modern convolutional neural networks are easier to optimize. Why is this the case? In this work, we conjecture that the issue lies with the patchify stem of ViT models, which is implemented by a stride-p p*p convolution (p=16 by default) applied to the input image. This large-kernel plus large-stride convolution runs counter to typical design choices of convolutional layers in ne"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2106.14881","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2106.14881/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2106.14881","created_at":"2026-07-05T03:25:32.496991+00:00"},{"alias_kind":"arxiv_version","alias_value":"2106.14881v3","created_at":"2026-07-05T03:25:32.496991+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2106.14881","created_at":"2026-07-05T03:25:32.496991+00:00"},{"alias_kind":"pith_short_12","alias_value":"XIUH7MZC4Q4W","created_at":"2026-07-05T03:25:32.496991+00:00"},{"alias_kind":"pith_short_16","alias_value":"XIUH7MZC4Q4WJMGT","created_at":"2026-07-05T03:25:32.496991+00:00"},{"alias_kind":"pith_short_8","alias_value":"XIUH7MZC","created_at":"2026-07-05T03:25:32.496991+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.06281","citing_title":"Multi-Resolution Tactile Imitation Learning for Contact-Rich Robotic Manipulation","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28796","citing_title":"Structure-Preserving Document Translation via Multi-Stage LLM Pipeline: A Case Study in Marathi","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2110.02178","citing_title":"MobileViT: Light-weight, General-purpose, and Mobile-friendly Vision Transformer","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2511.14427","citing_title":"Self-Supervised Multisensory Pretraining for Contact-Rich Robot Reinforcement Learning","ref_index":48,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XIUH7MZC4Q4WJMGTYYJ2AH4QRE","json":"https://pith.science/pith/XIUH7MZC4Q4WJMGTYYJ2AH4QRE.json","graph_json":"https://pith.science/api/pith-number/XIUH7MZC4Q4WJMGTYYJ2AH4QRE/graph.json","events_json":"https://pith.science/api/pith-number/XIUH7MZC4Q4WJMGTYYJ2AH4QRE/events.json","paper":"https://pith.science/paper/XIUH7MZC"},"agent_actions":{"view_html":"https://pith.science/pith/XIUH7MZC4Q4WJMGTYYJ2AH4QRE","download_json":"https://pith.science/pith/XIUH7MZC4Q4WJMGTYYJ2AH4QRE.json","view_paper":"https://pith.science/paper/XIUH7MZC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2106.14881&json=true","fetch_graph":"https://pith.science/api/pith-number/XIUH7MZC4Q4WJMGTYYJ2AH4QRE/graph.json","fetch_events":"https://pith.science/api/pith-number/XIUH7MZC4Q4WJMGTYYJ2AH4QRE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XIUH7MZC4Q4WJMGTYYJ2AH4QRE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XIUH7MZC4Q4WJMGTYYJ2AH4QRE/action/storage_attestation","attest_author":"https://pith.science/pith/XIUH7MZC4Q4WJMGTYYJ2AH4QRE/action/author_attestation","sign_citation":"https://pith.science/pith/XIUH7MZC4Q4WJMGTYYJ2AH4QRE/action/citation_signature","submit_replication":"https://pith.science/pith/XIUH7MZC4Q4WJMGTYYJ2AH4QRE/action/replication_record"}},"created_at":"2026-07-05T03:25:32.496991+00:00","updated_at":"2026-07-05T03:25:32.496991+00:00"}