{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:ZNTUZKU5MWETDOWXLEFXWMURSZ","short_pith_number":"pith:ZNTUZKU5","schema_version":"1.0","canonical_sha256":"cb674caa9d658931bad7590b7b3291964c735642fa78436ad935a339bc078c74","source":{"kind":"arxiv","id":"2308.09372","version":4},"attestation_state":"computed","paper":{"title":"Which Transformer to Favor: A Comparative Analysis of Efficiency in Vision Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Andreas Dengel, Federico Raue, Sebastian Palacio, Tobias Christian Nauen","submitted_at":"2023-08-18T08:06:49Z","abstract_excerpt":"Self-attention in Transformers comes with a high computational cost because of their quadratic computational complexity, but their effectiveness in addressing problems in language and vision has sparked extensive research aimed at enhancing their efficiency. However, diverse experimental conditions, spanning multiple input domains, prevent a fair comparison based solely on reported results, posing challenges for model selection. To address this gap in comparability, we perform a large-scale benchmark of more than 45 models for image classification, evaluating key efficiency aspects, including "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.09372","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-08-18T08:06:49Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"553c5226400cc897558f38b0bbee0f17c0ad0cd846d9b5eda11f16d830989e1c","abstract_canon_sha256":"a5aadedd19b8ce63e62b676929856561eebfa69294cd701cac8859fb4ce77f35"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:18:32.529873Z","signature_b64":"8xtO1CemiWeHjEV2cUcxEHzfzY/eXlzWCNUna3WoXJyKVVt9iOyNg1SmKANV9Gm55OkbR2xGClV/muVgk6jdDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cb674caa9d658931bad7590b7b3291964c735642fa78436ad935a339bc078c74","last_reissued_at":"2026-07-05T10:18:32.529410Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:18:32.529410Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Which Transformer to Favor: A Comparative Analysis of Efficiency in Vision Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Andreas Dengel, Federico Raue, Sebastian Palacio, Tobias Christian Nauen","submitted_at":"2023-08-18T08:06:49Z","abstract_excerpt":"Self-attention in Transformers comes with a high computational cost because of their quadratic computational complexity, but their effectiveness in addressing problems in language and vision has sparked extensive research aimed at enhancing their efficiency. However, diverse experimental conditions, spanning multiple input domains, prevent a fair comparison based solely on reported results, posing challenges for model selection. To address this gap in comparability, we perform a large-scale benchmark of more than 45 models for image classification, evaluating key efficiency aspects, including "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.09372","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.09372/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.09372","created_at":"2026-07-05T10:18:32.529468+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.09372v4","created_at":"2026-07-05T10:18:32.529468+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.09372","created_at":"2026-07-05T10:18:32.529468+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZNTUZKU5MWET","created_at":"2026-07-05T10:18:32.529468+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZNTUZKU5MWETDOWX","created_at":"2026-07-05T10:18:32.529468+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZNTUZKU5","created_at":"2026-07-05T10:18:32.529468+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.10893","citing_title":"Modernizing CNN-based Weather Forecast Model towards Higher Computational Efficiency","ref_index":6,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZNTUZKU5MWETDOWXLEFXWMURSZ","json":"https://pith.science/pith/ZNTUZKU5MWETDOWXLEFXWMURSZ.json","graph_json":"https://pith.science/api/pith-number/ZNTUZKU5MWETDOWXLEFXWMURSZ/graph.json","events_json":"https://pith.science/api/pith-number/ZNTUZKU5MWETDOWXLEFXWMURSZ/events.json","paper":"https://pith.science/paper/ZNTUZKU5"},"agent_actions":{"view_html":"https://pith.science/pith/ZNTUZKU5MWETDOWXLEFXWMURSZ","download_json":"https://pith.science/pith/ZNTUZKU5MWETDOWXLEFXWMURSZ.json","view_paper":"https://pith.science/paper/ZNTUZKU5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.09372&json=true","fetch_graph":"https://pith.science/api/pith-number/ZNTUZKU5MWETDOWXLEFXWMURSZ/graph.json","fetch_events":"https://pith.science/api/pith-number/ZNTUZKU5MWETDOWXLEFXWMURSZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZNTUZKU5MWETDOWXLEFXWMURSZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZNTUZKU5MWETDOWXLEFXWMURSZ/action/storage_attestation","attest_author":"https://pith.science/pith/ZNTUZKU5MWETDOWXLEFXWMURSZ/action/author_attestation","sign_citation":"https://pith.science/pith/ZNTUZKU5MWETDOWXLEFXWMURSZ/action/citation_signature","submit_replication":"https://pith.science/pith/ZNTUZKU5MWETDOWXLEFXWMURSZ/action/replication_record"}},"created_at":"2026-07-05T10:18:32.529468+00:00","updated_at":"2026-07-05T10:18:32.529468+00:00"}