{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:SCE7WBSZZVUD73FDGLDUY42ZBB","short_pith_number":"pith:SCE7WBSZ","schema_version":"1.0","canonical_sha256":"9089fb0659cd683feca332c74c7359087096226d491cb4b73f4491669309b1f9","source":{"kind":"arxiv","id":"2411.14429","version":1},"attestation_state":"computed","paper":{"title":"Revisiting the Integration of Convolution and Attention for Vision Backbone","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Lei Zhu, Rynson W. H. Lau, Wayne Zhang, Xinjiang Wang","submitted_at":"2024-11-21T18:59:08Z","abstract_excerpt":"Convolutions (Convs) and multi-head self-attentions (MHSAs) are typically considered alternatives to each other for building vision backbones. Although some works try to integrate both, they apply the two operators simultaneously at the finest pixel granularity. With Convs responsible for per-pixel feature extraction already, the question is whether we still need to include the heavy MHSAs at such a fine-grained level. In fact, this is the root cause of the scalability issue w.r.t. the input resolution for vision transformers. To address this important problem, we propose in this work to use M"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.14429","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-11-21T18:59:08Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"5869b8fbc16e3596c210dcfea881838a079dc6a8f56029818183625468453f11","abstract_canon_sha256":"c574963b13632f9f090ee1aba607d7720a72fa339b94bfe1807840f7bc66fc80"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:38:46.861332Z","signature_b64":"9BwFxatd498cJsYsLzSvoyUJyetezG4k/WC5w2wiQu74hHWQTQ5J6t9jeUPZkUosEMVF4Eb/gce5J9DJJcc9Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9089fb0659cd683feca332c74c7359087096226d491cb4b73f4491669309b1f9","last_reissued_at":"2026-07-05T09:38:46.860857Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:38:46.860857Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Revisiting the Integration of Convolution and Attention for Vision Backbone","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Lei Zhu, Rynson W. H. Lau, Wayne Zhang, Xinjiang Wang","submitted_at":"2024-11-21T18:59:08Z","abstract_excerpt":"Convolutions (Convs) and multi-head self-attentions (MHSAs) are typically considered alternatives to each other for building vision backbones. Although some works try to integrate both, they apply the two operators simultaneously at the finest pixel granularity. With Convs responsible for per-pixel feature extraction already, the question is whether we still need to include the heavy MHSAs at such a fine-grained level. In fact, this is the root cause of the scalability issue w.r.t. the input resolution for vision transformers. To address this important problem, we propose in this work to use M"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.14429","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.14429/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.14429","created_at":"2026-07-05T09:38:46.860913+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.14429v1","created_at":"2026-07-05T09:38:46.860913+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.14429","created_at":"2026-07-05T09:38:46.860913+00:00"},{"alias_kind":"pith_short_12","alias_value":"SCE7WBSZZVUD","created_at":"2026-07-05T09:38:46.860913+00:00"},{"alias_kind":"pith_short_16","alias_value":"SCE7WBSZZVUD73FD","created_at":"2026-07-05T09:38:46.860913+00:00"},{"alias_kind":"pith_short_8","alias_value":"SCE7WBSZ","created_at":"2026-07-05T09:38:46.860913+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SCE7WBSZZVUD73FDGLDUY42ZBB","json":"https://pith.science/pith/SCE7WBSZZVUD73FDGLDUY42ZBB.json","graph_json":"https://pith.science/api/pith-number/SCE7WBSZZVUD73FDGLDUY42ZBB/graph.json","events_json":"https://pith.science/api/pith-number/SCE7WBSZZVUD73FDGLDUY42ZBB/events.json","paper":"https://pith.science/paper/SCE7WBSZ"},"agent_actions":{"view_html":"https://pith.science/pith/SCE7WBSZZVUD73FDGLDUY42ZBB","download_json":"https://pith.science/pith/SCE7WBSZZVUD73FDGLDUY42ZBB.json","view_paper":"https://pith.science/paper/SCE7WBSZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.14429&json=true","fetch_graph":"https://pith.science/api/pith-number/SCE7WBSZZVUD73FDGLDUY42ZBB/graph.json","fetch_events":"https://pith.science/api/pith-number/SCE7WBSZZVUD73FDGLDUY42ZBB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SCE7WBSZZVUD73FDGLDUY42ZBB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SCE7WBSZZVUD73FDGLDUY42ZBB/action/storage_attestation","attest_author":"https://pith.science/pith/SCE7WBSZZVUD73FDGLDUY42ZBB/action/author_attestation","sign_citation":"https://pith.science/pith/SCE7WBSZZVUD73FDGLDUY42ZBB/action/citation_signature","submit_replication":"https://pith.science/pith/SCE7WBSZZVUD73FDGLDUY42ZBB/action/replication_record"}},"created_at":"2026-07-05T09:38:46.860913+00:00","updated_at":"2026-07-05T09:38:46.860913+00:00"}