{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:47PLXNJG5YGSMFKU4HBTUUROAK","short_pith_number":"pith:47PLXNJG","schema_version":"1.0","canonical_sha256":"e7debbb526ee0d261554e1c33a522e02b612183da72a87802f9af7eac44eeea4","source":{"kind":"arxiv","id":"2307.09283","version":8},"attestation_state":"computed","paper":{"title":"RepViT: Revisiting Mobile CNN From ViT Perspective","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Ao Wang, Guiguang Ding, Hui Chen, Jungong Han, Zijia Lin","submitted_at":"2023-07-18T14:24:33Z","abstract_excerpt":"Recently, lightweight Vision Transformers (ViTs) demonstrate superior performance and lower latency, compared with lightweight Convolutional Neural Networks (CNNs), on resource-constrained mobile devices. Researchers have discovered many structural connections between lightweight ViTs and lightweight CNNs. However, the notable architectural disparities in the block structure, macro, and micro designs between them have not been adequately examined. In this study, we revisit the efficient design of lightweight CNNs from ViT perspective and emphasize their promising prospect for mobile devices. S"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.09283","kind":"arxiv","version":8},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-07-18T14:24:33Z","cross_cats_sorted":[],"title_canon_sha256":"1dfecc755544390d9dcf1e25c0f2013bfaf6da3a25c6a78f0333866dae04f30c","abstract_canon_sha256":"cdcf3f9c651417546e33be7f7c3eacb510e85cb387a84d71793cb9a5e2b099fb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:55:58.582908Z","signature_b64":"MNMDRHnf+dG4IXtKjbPzJ6R4p+Whic8CJPf023pzSac6TxM7Lb4r2502QCANtTISJE7WzlCmfgD5Xsw6COlhBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e7debbb526ee0d261554e1c33a522e02b612183da72a87802f9af7eac44eeea4","last_reissued_at":"2026-07-05T07:55:58.582390Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:55:58.582390Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RepViT: Revisiting Mobile CNN From ViT Perspective","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Ao Wang, Guiguang Ding, Hui Chen, Jungong Han, Zijia Lin","submitted_at":"2023-07-18T14:24:33Z","abstract_excerpt":"Recently, lightweight Vision Transformers (ViTs) demonstrate superior performance and lower latency, compared with lightweight Convolutional Neural Networks (CNNs), on resource-constrained mobile devices. Researchers have discovered many structural connections between lightweight ViTs and lightweight CNNs. However, the notable architectural disparities in the block structure, macro, and micro designs between them have not been adequately examined. In this study, we revisit the efficient design of lightweight CNNs from ViT perspective and emphasize their promising prospect for mobile devices. S"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.09283","kind":"arxiv","version":8},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.09283/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.09283","created_at":"2026-07-05T07:55:58.582448+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.09283v8","created_at":"2026-07-05T07:55:58.582448+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.09283","created_at":"2026-07-05T07:55:58.582448+00:00"},{"alias_kind":"pith_short_12","alias_value":"47PLXNJG5YGS","created_at":"2026-07-05T07:55:58.582448+00:00"},{"alias_kind":"pith_short_16","alias_value":"47PLXNJG5YGSMFKU","created_at":"2026-07-05T07:55:58.582448+00:00"},{"alias_kind":"pith_short_8","alias_value":"47PLXNJG","created_at":"2026-07-05T07:55:58.582448+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.00746","citing_title":"Scaling Parallel Sequence Models to Foundation-Scale Vision Encoders","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2408.12974","citing_title":"Accuracy Improvement of Cell Image Segmentation Using Feedback Former","ref_index":27,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/47PLXNJG5YGSMFKU4HBTUUROAK","json":"https://pith.science/pith/47PLXNJG5YGSMFKU4HBTUUROAK.json","graph_json":"https://pith.science/api/pith-number/47PLXNJG5YGSMFKU4HBTUUROAK/graph.json","events_json":"https://pith.science/api/pith-number/47PLXNJG5YGSMFKU4HBTUUROAK/events.json","paper":"https://pith.science/paper/47PLXNJG"},"agent_actions":{"view_html":"https://pith.science/pith/47PLXNJG5YGSMFKU4HBTUUROAK","download_json":"https://pith.science/pith/47PLXNJG5YGSMFKU4HBTUUROAK.json","view_paper":"https://pith.science/paper/47PLXNJG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.09283&json=true","fetch_graph":"https://pith.science/api/pith-number/47PLXNJG5YGSMFKU4HBTUUROAK/graph.json","fetch_events":"https://pith.science/api/pith-number/47PLXNJG5YGSMFKU4HBTUUROAK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/47PLXNJG5YGSMFKU4HBTUUROAK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/47PLXNJG5YGSMFKU4HBTUUROAK/action/storage_attestation","attest_author":"https://pith.science/pith/47PLXNJG5YGSMFKU4HBTUUROAK/action/author_attestation","sign_citation":"https://pith.science/pith/47PLXNJG5YGSMFKU4HBTUUROAK/action/citation_signature","submit_replication":"https://pith.science/pith/47PLXNJG5YGSMFKU4HBTUUROAK/action/replication_record"}},"created_at":"2026-07-05T07:55:58.582448+00:00","updated_at":"2026-07-05T07:55:58.582448+00:00"}