{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:BS7W4L7WSHKYPG3NJXMOGSWUTA","short_pith_number":"pith:BS7W4L7W","schema_version":"1.0","canonical_sha256":"0cbf6e2ff691d5879b6d4dd8e34ad4982363837a7cb8b1cbc1f34d73fbfda43c","source":{"kind":"arxiv","id":"2403.09338","version":1},"attestation_state":"computed","paper":{"title":"LocalMamba: Visual State Space Model with Windowed Selective Scan","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Chang Xu, Chen Qian, Fei Wang, Shan You, Tao Huang, Xiaohuan Pei","submitted_at":"2024-03-14T12:32:40Z","abstract_excerpt":"Recent advancements in state space models, notably Mamba, have demonstrated significant progress in modeling long sequences for tasks like language understanding. Yet, their application in vision tasks has not markedly surpassed the performance of traditional Convolutional Neural Networks (CNNs) and Vision Transformers (ViTs). This paper posits that the key to enhancing Vision Mamba (ViM) lies in optimizing scan directions for sequence modeling. Traditional ViM approaches, which flatten spatial tokens, overlook the preservation of local 2D dependencies, thereby elongating the distance between "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.09338","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-03-14T12:32:40Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"ee01ce653d8ab186e4e83f79d088454da9c713a6029cafd0386e04a45573f3df","abstract_canon_sha256":"8d999ba75df25ab20861d9c5563a14839a88e3dabc3a649ff2bd40599f838971"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:56:08.789822Z","signature_b64":"7pzvQgTYePThyJ9mgTxrXyaQKfPtY5STldpGnXozX2KWBfDHuetxHIs0n0NOn8pj6DL/ro4Qh8BI8CMfy0iDAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0cbf6e2ff691d5879b6d4dd8e34ad4982363837a7cb8b1cbc1f34d73fbfda43c","last_reissued_at":"2026-07-05T07:56:08.789349Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:56:08.789349Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LocalMamba: Visual State Space Model with Windowed Selective Scan","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Chang Xu, Chen Qian, Fei Wang, Shan You, Tao Huang, Xiaohuan Pei","submitted_at":"2024-03-14T12:32:40Z","abstract_excerpt":"Recent advancements in state space models, notably Mamba, have demonstrated significant progress in modeling long sequences for tasks like language understanding. Yet, their application in vision tasks has not markedly surpassed the performance of traditional Convolutional Neural Networks (CNNs) and Vision Transformers (ViTs). This paper posits that the key to enhancing Vision Mamba (ViM) lies in optimizing scan directions for sequence modeling. Traditional ViM approaches, which flatten spatial tokens, overlook the preservation of local 2D dependencies, thereby elongating the distance between "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.09338","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.09338/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.09338","created_at":"2026-07-05T07:56:08.789413+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.09338v1","created_at":"2026-07-05T07:56:08.789413+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.09338","created_at":"2026-07-05T07:56:08.789413+00:00"},{"alias_kind":"pith_short_12","alias_value":"BS7W4L7WSHKY","created_at":"2026-07-05T07:56:08.789413+00:00"},{"alias_kind":"pith_short_16","alias_value":"BS7W4L7WSHKYPG3N","created_at":"2026-07-05T07:56:08.789413+00:00"},{"alias_kind":"pith_short_8","alias_value":"BS7W4L7W","created_at":"2026-07-05T07:56:08.789413+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23126","citing_title":"MambaADv2: Evolving Duality-enhanced State Space Model for Unsupervised Anomaly Detection","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17966","citing_title":"Reload-Mamba: Hierarchical Anti-Dilution State-Space Modeling for Multi-Class Semantic Segmentation","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14799","citing_title":"Can Visual Mamba Improve AI-Generated Image Detection? An In-Depth Investigation","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00746","citing_title":"Scaling Parallel Sequence Models to Foundation-Scale Vision Encoders","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2408.01129","citing_title":"A Survey of Mamba","ref_index":81,"is_internal_anchor":false},{"citing_arxiv_id":"2501.15461","citing_title":"Mamba-Based Graph Convolutional Networks: Tackling Over-smoothing with Selective State Space","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2505.14062","citing_title":"FractalMamba++: Scaling Vision Mamba Across Resolutions via Hilbert Fractal Geometry","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12640","citing_title":"MambaPanoptic: A Vision Mamba-based Structured State Space Framework for Panoptic Segmentation","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12640","citing_title":"MambaPanoptic: A Vision Mamba-based Structured State Space Framework for Panoptic Segmentation","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14724","citing_title":"HAMSA: Scanning-Free Vision State Space Models via SpectralPulseNet","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20606","citing_title":"Beyond ZOH: Advanced Discretization Strategies for Vision Mamba","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BS7W4L7WSHKYPG3NJXMOGSWUTA","json":"https://pith.science/pith/BS7W4L7WSHKYPG3NJXMOGSWUTA.json","graph_json":"https://pith.science/api/pith-number/BS7W4L7WSHKYPG3NJXMOGSWUTA/graph.json","events_json":"https://pith.science/api/pith-number/BS7W4L7WSHKYPG3NJXMOGSWUTA/events.json","paper":"https://pith.science/paper/BS7W4L7W"},"agent_actions":{"view_html":"https://pith.science/pith/BS7W4L7WSHKYPG3NJXMOGSWUTA","download_json":"https://pith.science/pith/BS7W4L7WSHKYPG3NJXMOGSWUTA.json","view_paper":"https://pith.science/paper/BS7W4L7W","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.09338&json=true","fetch_graph":"https://pith.science/api/pith-number/BS7W4L7WSHKYPG3NJXMOGSWUTA/graph.json","fetch_events":"https://pith.science/api/pith-number/BS7W4L7WSHKYPG3NJXMOGSWUTA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BS7W4L7WSHKYPG3NJXMOGSWUTA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BS7W4L7WSHKYPG3NJXMOGSWUTA/action/storage_attestation","attest_author":"https://pith.science/pith/BS7W4L7WSHKYPG3NJXMOGSWUTA/action/author_attestation","sign_citation":"https://pith.science/pith/BS7W4L7WSHKYPG3NJXMOGSWUTA/action/citation_signature","submit_replication":"https://pith.science/pith/BS7W4L7WSHKYPG3NJXMOGSWUTA/action/replication_record"}},"created_at":"2026-07-05T07:56:08.789413+00:00","updated_at":"2026-07-05T07:56:08.789413+00:00"}