{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:LFOBVSFVYY5PWRD4WDHX2PKM4Y","short_pith_number":"pith:LFOBVSFV","schema_version":"1.0","canonical_sha256":"595c1ac8b5c63afb447cb0cf7d3d4ce61d1ade956a91e127db9f9c7571cc6d85","source":{"kind":"arxiv","id":"2108.13002","version":2},"attestation_state":"computed","paper":{"title":"A Battle of Network Structures: An Empirical Study of CNN, Transformer, and MLP","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chong Luo, Chuanxin Tang, Guangting Wang, Wenjun Zeng, Yucheng Zhao, Zheng-Jun Zha","submitted_at":"2021-08-30T06:09:02Z","abstract_excerpt":"Convolutional neural networks (CNN) are the dominant deep neural network (DNN) architecture for computer vision. Recently, Transformer and multi-layer perceptron (MLP)-based models, such as Vision Transformer and MLP-Mixer, started to lead new trends as they showed promising results in the ImageNet classification task. In this paper, we conduct empirical studies on these DNN structures and try to understand their respective pros and cons. To ensure a fair comparison, we first develop a unified framework called SPACH which adopts separate modules for spatial and channel processing. Our experime"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2108.13002","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CV","submitted_at":"2021-08-30T06:09:02Z","cross_cats_sorted":[],"title_canon_sha256":"facec8cf77002466d80f067d4f8eed43e57ec6017a85c4b2328c006522a8d041","abstract_canon_sha256":"2623a093d9cfd22dc6db7ec739256f23a572a0834085775e9114199f431db9dd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:35:12.637628Z","signature_b64":"3RA1DWsMeFjXbiqlGKbsfrCNolEk98Rg4JHa9MU6TE9hur/6O6clmZ/v91zSOnCp+vabUoOuRNY+vfDVvl0nBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"595c1ac8b5c63afb447cb0cf7d3d4ce61d1ade956a91e127db9f9c7571cc6d85","last_reissued_at":"2026-07-05T03:35:12.637089Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:35:12.637089Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Battle of Network Structures: An Empirical Study of CNN, Transformer, and MLP","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chong Luo, Chuanxin Tang, Guangting Wang, Wenjun Zeng, Yucheng Zhao, Zheng-Jun Zha","submitted_at":"2021-08-30T06:09:02Z","abstract_excerpt":"Convolutional neural networks (CNN) are the dominant deep neural network (DNN) architecture for computer vision. Recently, Transformer and multi-layer perceptron (MLP)-based models, such as Vision Transformer and MLP-Mixer, started to lead new trends as they showed promising results in the ImageNet classification task. In this paper, we conduct empirical studies on these DNN structures and try to understand their respective pros and cons. To ensure a fair comparison, we first develop a unified framework called SPACH which adopts separate modules for spatial and channel processing. Our experime"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2108.13002","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2108.13002/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2108.13002","created_at":"2026-07-05T03:35:12.637152+00:00"},{"alias_kind":"arxiv_version","alias_value":"2108.13002v2","created_at":"2026-07-05T03:35:12.637152+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2108.13002","created_at":"2026-07-05T03:35:12.637152+00:00"},{"alias_kind":"pith_short_12","alias_value":"LFOBVSFVYY5P","created_at":"2026-07-05T03:35:12.637152+00:00"},{"alias_kind":"pith_short_16","alias_value":"LFOBVSFVYY5PWRD4","created_at":"2026-07-05T03:35:12.637152+00:00"},{"alias_kind":"pith_short_8","alias_value":"LFOBVSFV","created_at":"2026-07-05T03:35:12.637152+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.05586","citing_title":"BMCR: Adaptive Backbone Module Composition via Reinforcement Learning for Remote Sensing Object Detection","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2503.23947","citing_title":"Spectral-Adaptive Modulation Networks for Visual Perception","ref_index":48,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LFOBVSFVYY5PWRD4WDHX2PKM4Y","json":"https://pith.science/pith/LFOBVSFVYY5PWRD4WDHX2PKM4Y.json","graph_json":"https://pith.science/api/pith-number/LFOBVSFVYY5PWRD4WDHX2PKM4Y/graph.json","events_json":"https://pith.science/api/pith-number/LFOBVSFVYY5PWRD4WDHX2PKM4Y/events.json","paper":"https://pith.science/paper/LFOBVSFV"},"agent_actions":{"view_html":"https://pith.science/pith/LFOBVSFVYY5PWRD4WDHX2PKM4Y","download_json":"https://pith.science/pith/LFOBVSFVYY5PWRD4WDHX2PKM4Y.json","view_paper":"https://pith.science/paper/LFOBVSFV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2108.13002&json=true","fetch_graph":"https://pith.science/api/pith-number/LFOBVSFVYY5PWRD4WDHX2PKM4Y/graph.json","fetch_events":"https://pith.science/api/pith-number/LFOBVSFVYY5PWRD4WDHX2PKM4Y/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LFOBVSFVYY5PWRD4WDHX2PKM4Y/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LFOBVSFVYY5PWRD4WDHX2PKM4Y/action/storage_attestation","attest_author":"https://pith.science/pith/LFOBVSFVYY5PWRD4WDHX2PKM4Y/action/author_attestation","sign_citation":"https://pith.science/pith/LFOBVSFVYY5PWRD4WDHX2PKM4Y/action/citation_signature","submit_replication":"https://pith.science/pith/LFOBVSFVYY5PWRD4WDHX2PKM4Y/action/replication_record"}},"created_at":"2026-07-05T03:35:12.637152+00:00","updated_at":"2026-07-05T03:35:12.637152+00:00"}