{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:PDDEZTGGR5FKP4OF3JMGHXG2U5","short_pith_number":"pith:PDDEZTGG","schema_version":"1.0","canonical_sha256":"78c64cccc68f4aa7f1c5da5863dcdaa756fd634465781966d8e029576c87c396","source":{"kind":"arxiv","id":"2207.05501","version":4},"attestation_state":"computed","paper":{"title":"Next-ViT: Next Generation Vision Transformer for Efficient Deployment in Realistic Industrial Scenarios","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Huixia Li, Jiashi Li, Min Zheng, Rui Wang, Wei Li, Xing Wang, Xin Pan, Xin Xia, Xuefeng Xiao","submitted_at":"2022-07-12T12:50:34Z","abstract_excerpt":"Due to the complex attention mechanisms and model design, most existing vision Transformers (ViTs) can not perform as efficiently as convolutional neural networks (CNNs) in realistic industrial deployment scenarios, e.g. TensorRT and CoreML. This poses a distinct challenge: Can a visual neural network be designed to infer as fast as CNNs and perform as powerful as ViTs? Recent works have tried to design CNN-Transformer hybrid architectures to address this issue, yet the overall performance of these works is far away from satisfactory. To end these, we propose a next generation vision Transform"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2207.05501","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2022-07-12T12:50:34Z","cross_cats_sorted":[],"title_canon_sha256":"c7c1a55e58a5f811c982f31486c80d56f92caf02810b916f03bdd480e6e101d6","abstract_canon_sha256":"17590fc2ef9c24197d021acf82134ee11ecf65c5b86cacb233bcfb58db083272"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:48:58.707900Z","signature_b64":"4OuJ32gzzhvjxNFQAw12APv+RH2QhvqP91fI3RgWR+SVEI5o8g0sVBskzpDSJGQu8H5RdzaD/W69Y5wFUZdCAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"78c64cccc68f4aa7f1c5da5863dcdaa756fd634465781966d8e029576c87c396","last_reissued_at":"2026-07-05T04:48:58.707488Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:48:58.707488Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Next-ViT: Next Generation Vision Transformer for Efficient Deployment in Realistic Industrial Scenarios","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Huixia Li, Jiashi Li, Min Zheng, Rui Wang, Wei Li, Xing Wang, Xin Pan, Xin Xia, Xuefeng Xiao","submitted_at":"2022-07-12T12:50:34Z","abstract_excerpt":"Due to the complex attention mechanisms and model design, most existing vision Transformers (ViTs) can not perform as efficiently as convolutional neural networks (CNNs) in realistic industrial deployment scenarios, e.g. TensorRT and CoreML. This poses a distinct challenge: Can a visual neural network be designed to infer as fast as CNNs and perform as powerful as ViTs? Recent works have tried to design CNN-Transformer hybrid architectures to address this issue, yet the overall performance of these works is far away from satisfactory. To end these, we propose a next generation vision Transform"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2207.05501","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2207.05501/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2207.05501","created_at":"2026-07-05T04:48:58.707545+00:00"},{"alias_kind":"arxiv_version","alias_value":"2207.05501v4","created_at":"2026-07-05T04:48:58.707545+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2207.05501","created_at":"2026-07-05T04:48:58.707545+00:00"},{"alias_kind":"pith_short_12","alias_value":"PDDEZTGGR5FK","created_at":"2026-07-05T04:48:58.707545+00:00"},{"alias_kind":"pith_short_16","alias_value":"PDDEZTGGR5FKP4OF","created_at":"2026-07-05T04:48:58.707545+00:00"},{"alias_kind":"pith_short_8","alias_value":"PDDEZTGG","created_at":"2026-07-05T04:48:58.707545+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.00746","citing_title":"Scaling Parallel Sequence Models to Foundation-Scale Vision Encoders","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22098","citing_title":"TextTeacher: What Can Language Teach About Images?","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2306.14289","citing_title":"Faster Segment Anything: Towards Lightweight SAM for Mobile Applications","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06783","citing_title":"Insights from Visual Cognition: Understanding Human Action Dynamics with Overall Glance and Refined Gaze Transformer","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07338","citing_title":"ShellfishNet: A Domain-Specific Benchmark for Visual Recognition of Marine Molluscs","ref_index":52,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PDDEZTGGR5FKP4OF3JMGHXG2U5","json":"https://pith.science/pith/PDDEZTGGR5FKP4OF3JMGHXG2U5.json","graph_json":"https://pith.science/api/pith-number/PDDEZTGGR5FKP4OF3JMGHXG2U5/graph.json","events_json":"https://pith.science/api/pith-number/PDDEZTGGR5FKP4OF3JMGHXG2U5/events.json","paper":"https://pith.science/paper/PDDEZTGG"},"agent_actions":{"view_html":"https://pith.science/pith/PDDEZTGGR5FKP4OF3JMGHXG2U5","download_json":"https://pith.science/pith/PDDEZTGGR5FKP4OF3JMGHXG2U5.json","view_paper":"https://pith.science/paper/PDDEZTGG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2207.05501&json=true","fetch_graph":"https://pith.science/api/pith-number/PDDEZTGGR5FKP4OF3JMGHXG2U5/graph.json","fetch_events":"https://pith.science/api/pith-number/PDDEZTGGR5FKP4OF3JMGHXG2U5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PDDEZTGGR5FKP4OF3JMGHXG2U5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PDDEZTGGR5FKP4OF3JMGHXG2U5/action/storage_attestation","attest_author":"https://pith.science/pith/PDDEZTGGR5FKP4OF3JMGHXG2U5/action/author_attestation","sign_citation":"https://pith.science/pith/PDDEZTGGR5FKP4OF3JMGHXG2U5/action/citation_signature","submit_replication":"https://pith.science/pith/PDDEZTGGR5FKP4OF3JMGHXG2U5/action/replication_record"}},"created_at":"2026-07-05T04:48:58.707545+00:00","updated_at":"2026-07-05T04:48:58.707545+00:00"}