{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:YZW2HQRQUNHFZMI55FECHWFWWX","short_pith_number":"pith:YZW2HQRQ","schema_version":"1.0","canonical_sha256":"c66da3c230a34e5cb11de94823d8b6b5e0883ae2b4595b9f59a1be029d3f8102","source":{"kind":"arxiv","id":"2201.09792","version":1},"attestation_state":"computed","paper":{"title":"Patches Are All You Need?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Asher Trockman, J. Zico Kolter","submitted_at":"2022-01-24T16:42:56Z","abstract_excerpt":"Although convolutional networks have been the dominant architecture for vision tasks for many years, recent experiments have shown that Transformer-based models, most notably the Vision Transformer (ViT), may exceed their performance in some settings. However, due to the quadratic runtime of the self-attention layers in Transformers, ViTs require the use of patch embeddings, which group together small regions of the image into single input features, in order to be applied to larger image sizes. This raises a question: Is the performance of ViTs due to the inherently-more-powerful Transformer a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2201.09792","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2022-01-24T16:42:56Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"8567daadc49a7ddc11883e744bec8d6920046d8e77e921c2b8cd57752f0a6772","abstract_canon_sha256":"25c1f33ef09dc8f3f257715e1ba195dc172b6d9f8b88d1d20fc5ff662e206276"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:51:01.274423Z","signature_b64":"F1CMOEb+eZSyNsPO/qY7I/D891JuU9sismXN2S+5+MhKKIUYqqjxo98DNjjwhHsaH4oD51uUxxTCOC6/lZJWDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c66da3c230a34e5cb11de94823d8b6b5e0883ae2b4595b9f59a1be029d3f8102","last_reissued_at":"2026-07-05T03:51:01.273901Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:51:01.273901Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Patches Are All You Need?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Asher Trockman, J. Zico Kolter","submitted_at":"2022-01-24T16:42:56Z","abstract_excerpt":"Although convolutional networks have been the dominant architecture for vision tasks for many years, recent experiments have shown that Transformer-based models, most notably the Vision Transformer (ViT), may exceed their performance in some settings. However, due to the quadratic runtime of the self-attention layers in Transformers, ViTs require the use of patch embeddings, which group together small regions of the image into single input features, in order to be applied to larger image sizes. This raises a question: Is the performance of ViTs due to the inherently-more-powerful Transformer a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2201.09792","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2201.09792/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2201.09792","created_at":"2026-07-05T03:51:01.273959+00:00"},{"alias_kind":"arxiv_version","alias_value":"2201.09792v1","created_at":"2026-07-05T03:51:01.273959+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2201.09792","created_at":"2026-07-05T03:51:01.273959+00:00"},{"alias_kind":"pith_short_12","alias_value":"YZW2HQRQUNHF","created_at":"2026-07-05T03:51:01.273959+00:00"},{"alias_kind":"pith_short_16","alias_value":"YZW2HQRQUNHFZMI5","created_at":"2026-07-05T03:51:01.273959+00:00"},{"alias_kind":"pith_short_8","alias_value":"YZW2HQRQ","created_at":"2026-07-05T03:51:01.273959+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.02092","citing_title":"LALE: Lightweight-Transformer Architecture for Land-Cover Estimation","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03384","citing_title":"DECKER: Domain-invariant Embedding for Cross-Keyboard Extraction and Recognition","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2507.22101","citing_title":"AI in Agriculture: A Survey of Deep Learning Techniques for Crops, Fisheries and Livestock","ref_index":245,"is_internal_anchor":false},{"citing_arxiv_id":"2402.02366","citing_title":"Transolver: A Fast Transformer Solver for PDEs on General Geometries","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03384","citing_title":"DECKER: Domain-invariant Embedding for Cross-Keyboard Extraction and Recognition","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18067","citing_title":"Towards Real-Time ECG and EMG Modeling on $\\mu$NPUs","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03045","citing_title":"TCD-Arena: Assessing Robustness of Time Series Causal Discovery Methods Against Assumption Violations","ref_index":269,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YZW2HQRQUNHFZMI55FECHWFWWX","json":"https://pith.science/pith/YZW2HQRQUNHFZMI55FECHWFWWX.json","graph_json":"https://pith.science/api/pith-number/YZW2HQRQUNHFZMI55FECHWFWWX/graph.json","events_json":"https://pith.science/api/pith-number/YZW2HQRQUNHFZMI55FECHWFWWX/events.json","paper":"https://pith.science/paper/YZW2HQRQ"},"agent_actions":{"view_html":"https://pith.science/pith/YZW2HQRQUNHFZMI55FECHWFWWX","download_json":"https://pith.science/pith/YZW2HQRQUNHFZMI55FECHWFWWX.json","view_paper":"https://pith.science/paper/YZW2HQRQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2201.09792&json=true","fetch_graph":"https://pith.science/api/pith-number/YZW2HQRQUNHFZMI55FECHWFWWX/graph.json","fetch_events":"https://pith.science/api/pith-number/YZW2HQRQUNHFZMI55FECHWFWWX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YZW2HQRQUNHFZMI55FECHWFWWX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YZW2HQRQUNHFZMI55FECHWFWWX/action/storage_attestation","attest_author":"https://pith.science/pith/YZW2HQRQUNHFZMI55FECHWFWWX/action/author_attestation","sign_citation":"https://pith.science/pith/YZW2HQRQUNHFZMI55FECHWFWWX/action/citation_signature","submit_replication":"https://pith.science/pith/YZW2HQRQUNHFZMI55FECHWFWWX/action/replication_record"}},"created_at":"2026-07-05T03:51:01.273959+00:00","updated_at":"2026-07-05T03:51:01.273959+00:00"}