{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:4KYARUONCIF4MSX7UAQPN5XLM4","short_pith_number":"pith:4KYARUON","schema_version":"1.0","canonical_sha256":"e2b008d1cd120bc64affa020f6f6eb6715941919ef8bb64436abcc09b17979cf","source":{"kind":"arxiv","id":"2405.07992","version":3},"attestation_state":"computed","paper":{"title":"MambaOut: Do We Really Need Mamba for Vision?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Weihao Yu, Xinchao Wang","submitted_at":"2024-05-13T17:59:56Z","abstract_excerpt":"Mamba, an architecture with RNN-like token mixer of state space model (SSM), was recently introduced to address the quadratic complexity of the attention mechanism and subsequently applied to vision tasks. Nevertheless, the performance of Mamba for vision is often underwhelming when compared with convolutional and attention-based models. In this paper, we delve into the essence of Mamba, and conceptually conclude that Mamba is ideally suited for tasks with long-sequence and autoregressive characteristics. For vision tasks, as image classification does not align with either characteristic, we h"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.07992","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-05-13T17:59:56Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"40f2e1770dd488b0a6dcb3a7d9646dff69b9680c29153985e6440edd7848709c","abstract_canon_sha256":"322e1011029b13d063e57680c884891f69107db676118ca2429d43ce18f2eda6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:20:54.618396Z","signature_b64":"PnMiytTDOjyI8rwp0MCZRLXeEN9p3Cuiwws8YJdthm07lgtFV9LCBD+ZKpWq6JxMVDt15zpcM2d3VPtaqFwQBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e2b008d1cd120bc64affa020f6f6eb6715941919ef8bb64436abcc09b17979cf","last_reissued_at":"2026-07-05T08:20:54.617824Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:20:54.617824Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MambaOut: Do We Really Need Mamba for Vision?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Weihao Yu, Xinchao Wang","submitted_at":"2024-05-13T17:59:56Z","abstract_excerpt":"Mamba, an architecture with RNN-like token mixer of state space model (SSM), was recently introduced to address the quadratic complexity of the attention mechanism and subsequently applied to vision tasks. Nevertheless, the performance of Mamba for vision is often underwhelming when compared with convolutional and attention-based models. In this paper, we delve into the essence of Mamba, and conceptually conclude that Mamba is ideally suited for tasks with long-sequence and autoregressive characteristics. For vision tasks, as image classification does not align with either characteristic, we h"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.07992","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.07992/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.07992","created_at":"2026-07-05T08:20:54.617885+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.07992v3","created_at":"2026-07-05T08:20:54.617885+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.07992","created_at":"2026-07-05T08:20:54.617885+00:00"},{"alias_kind":"pith_short_12","alias_value":"4KYARUONCIF4","created_at":"2026-07-05T08:20:54.617885+00:00"},{"alias_kind":"pith_short_16","alias_value":"4KYARUONCIF4MSX7","created_at":"2026-07-05T08:20:54.617885+00:00"},{"alias_kind":"pith_short_8","alias_value":"4KYARUON","created_at":"2026-07-05T08:20:54.617885+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.00746","citing_title":"Scaling Parallel Sequence Models to Foundation-Scale Vision Encoders","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2512.07078","citing_title":"DFIR-DETR: Frequency-Domain Iterative Refinement and Dynamic Feature Aggregation for Small Object Detection","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2509.23310","citing_title":"Balanced Diffusion-Guided Fusion for Multimodal Remote Sensing Classification","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2510.04628","citing_title":"A Spatial-Spectral-Frequency Interactive Network for Multimodal Remote Sensing Classification","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13202","citing_title":"STAR: Semantic-Temporal Adaptive Representation Learning for Few-Shot Action Recognition","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01667","citing_title":"Deep neural networks with Fisher vector encoding for medical image classification","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14724","citing_title":"HAMSA: Scanning-Free Vision State Space Models via SpectralPulseNet","ref_index":74,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02794","citing_title":"Edge-Efficient Image Restoration: Transformer Distillation into State-Space Models","ref_index":63,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4KYARUONCIF4MSX7UAQPN5XLM4","json":"https://pith.science/pith/4KYARUONCIF4MSX7UAQPN5XLM4.json","graph_json":"https://pith.science/api/pith-number/4KYARUONCIF4MSX7UAQPN5XLM4/graph.json","events_json":"https://pith.science/api/pith-number/4KYARUONCIF4MSX7UAQPN5XLM4/events.json","paper":"https://pith.science/paper/4KYARUON"},"agent_actions":{"view_html":"https://pith.science/pith/4KYARUONCIF4MSX7UAQPN5XLM4","download_json":"https://pith.science/pith/4KYARUONCIF4MSX7UAQPN5XLM4.json","view_paper":"https://pith.science/paper/4KYARUON","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.07992&json=true","fetch_graph":"https://pith.science/api/pith-number/4KYARUONCIF4MSX7UAQPN5XLM4/graph.json","fetch_events":"https://pith.science/api/pith-number/4KYARUONCIF4MSX7UAQPN5XLM4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4KYARUONCIF4MSX7UAQPN5XLM4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4KYARUONCIF4MSX7UAQPN5XLM4/action/storage_attestation","attest_author":"https://pith.science/pith/4KYARUONCIF4MSX7UAQPN5XLM4/action/author_attestation","sign_citation":"https://pith.science/pith/4KYARUONCIF4MSX7UAQPN5XLM4/action/citation_signature","submit_replication":"https://pith.science/pith/4KYARUONCIF4MSX7UAQPN5XLM4/action/replication_record"}},"created_at":"2026-07-05T08:20:54.617885+00:00","updated_at":"2026-07-05T08:20:54.617885+00:00"}