{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:N3CXAIBOLYRO74MGFTI4BHXK53","short_pith_number":"pith:N3CXAIBO","schema_version":"1.0","canonical_sha256":"6ec570202e5e22eff1862cd1c09eeaeedf4d3c040184ab15ee8d9d00c702f43b","source":{"kind":"arxiv","id":"2107.00651","version":1},"attestation_state":"computed","paper":{"title":"AutoFormer: Searching Transformers for Visual Recognition","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Haibin Ling, Houwen Peng, Jianlong Fu, Minghao Chen","submitted_at":"2021-07-01T17:59:30Z","abstract_excerpt":"Recently, pure transformer-based models have shown great potentials for vision tasks such as image classification and detection. However, the design of transformer networks is challenging. It has been observed that the depth, embedding dimension, and number of heads can largely affect the performance of vision transformers. Previous models configure these dimensions based upon manual crafting. In this work, we propose a new one-shot architecture search framework, namely AutoFormer, dedicated to vision transformer search. AutoFormer entangles the weights of different blocks in the same layers d"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2107.00651","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2021-07-01T17:59:30Z","cross_cats_sorted":[],"title_canon_sha256":"d7cc741aa1ae2be02e02698e8405b2a813cb7edca58804134ee37acddd43898b","abstract_canon_sha256":"116a74bc6eb769bc396b238896f29ea44b398a3e397c281d8164e7a1fa004bdd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:54:26.783286Z","signature_b64":"KYAbggknMgekAhAzhsNbpBTlVelvXGLNFbwCujEk47ouDQOXfztvtvAhkrjxftYZU/wtv/YjLIl7BOD9yBXCCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6ec570202e5e22eff1862cd1c09eeaeedf4d3c040184ab15ee8d9d00c702f43b","last_reissued_at":"2026-07-05T02:54:26.782852Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:54:26.782852Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AutoFormer: Searching Transformers for Visual Recognition","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Haibin Ling, Houwen Peng, Jianlong Fu, Minghao Chen","submitted_at":"2021-07-01T17:59:30Z","abstract_excerpt":"Recently, pure transformer-based models have shown great potentials for vision tasks such as image classification and detection. However, the design of transformer networks is challenging. It has been observed that the depth, embedding dimension, and number of heads can largely affect the performance of vision transformers. Previous models configure these dimensions based upon manual crafting. In this work, we propose a new one-shot architecture search framework, namely AutoFormer, dedicated to vision transformer search. AutoFormer entangles the weights of different blocks in the same layers d"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2107.00651","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2107.00651/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2107.00651","created_at":"2026-07-05T02:54:26.782909+00:00"},{"alias_kind":"arxiv_version","alias_value":"2107.00651v1","created_at":"2026-07-05T02:54:26.782909+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2107.00651","created_at":"2026-07-05T02:54:26.782909+00:00"},{"alias_kind":"pith_short_12","alias_value":"N3CXAIBOLYRO","created_at":"2026-07-05T02:54:26.782909+00:00"},{"alias_kind":"pith_short_16","alias_value":"N3CXAIBOLYRO74MG","created_at":"2026-07-05T02:54:26.782909+00:00"},{"alias_kind":"pith_short_8","alias_value":"N3CXAIBO","created_at":"2026-07-05T02:54:26.782909+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/N3CXAIBOLYRO74MGFTI4BHXK53","json":"https://pith.science/pith/N3CXAIBOLYRO74MGFTI4BHXK53.json","graph_json":"https://pith.science/api/pith-number/N3CXAIBOLYRO74MGFTI4BHXK53/graph.json","events_json":"https://pith.science/api/pith-number/N3CXAIBOLYRO74MGFTI4BHXK53/events.json","paper":"https://pith.science/paper/N3CXAIBO"},"agent_actions":{"view_html":"https://pith.science/pith/N3CXAIBOLYRO74MGFTI4BHXK53","download_json":"https://pith.science/pith/N3CXAIBOLYRO74MGFTI4BHXK53.json","view_paper":"https://pith.science/paper/N3CXAIBO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2107.00651&json=true","fetch_graph":"https://pith.science/api/pith-number/N3CXAIBOLYRO74MGFTI4BHXK53/graph.json","fetch_events":"https://pith.science/api/pith-number/N3CXAIBOLYRO74MGFTI4BHXK53/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/N3CXAIBOLYRO74MGFTI4BHXK53/action/timestamp_anchor","attest_storage":"https://pith.science/pith/N3CXAIBOLYRO74MGFTI4BHXK53/action/storage_attestation","attest_author":"https://pith.science/pith/N3CXAIBOLYRO74MGFTI4BHXK53/action/author_attestation","sign_citation":"https://pith.science/pith/N3CXAIBOLYRO74MGFTI4BHXK53/action/citation_signature","submit_replication":"https://pith.science/pith/N3CXAIBOLYRO74MGFTI4BHXK53/action/replication_record"}},"created_at":"2026-07-05T02:54:26.782909+00:00","updated_at":"2026-07-05T02:54:26.782909+00:00"}