{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:TRMWVJGO4EWTZNDVWMKG2M6FGU","short_pith_number":"pith:TRMWVJGO","canonical_record":{"source":{"id":"2401.09417","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-01-17T18:56:18Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"8014b4e743e657cb6756a563aefd1361d584179b6474d0d0c0594fbe86770470","abstract_canon_sha256":"08893014a214076870d708961c8371b9aebd84ca543d8ef36511c9d76d677895"},"schema_version":"1.0"},"canonical_sha256":"9c596aa4cee12d3cb475b3146d33c5350faa4378cd766283cfb0195a589f2a3c","source":{"kind":"arxiv","id":"2401.09417","version":3},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2401.09417","created_at":"2026-07-05T09:35:07Z"},{"alias_kind":"arxiv_version","alias_value":"2401.09417v3","created_at":"2026-07-05T09:35:07Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.09417","created_at":"2026-07-05T09:35:07Z"},{"alias_kind":"pith_short_12","alias_value":"TRMWVJGO4EWT","created_at":"2026-07-05T09:35:07Z"},{"alias_kind":"pith_short_16","alias_value":"TRMWVJGO4EWTZNDV","created_at":"2026-07-05T09:35:07Z"},{"alias_kind":"pith_short_8","alias_value":"TRMWVJGO","created_at":"2026-07-05T09:35:07Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:TRMWVJGO4EWTZNDVWMKG2M6FGU","target":"record","payload":{"canonical_record":{"source":{"id":"2401.09417","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-01-17T18:56:18Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"8014b4e743e657cb6756a563aefd1361d584179b6474d0d0c0594fbe86770470","abstract_canon_sha256":"08893014a214076870d708961c8371b9aebd84ca543d8ef36511c9d76d677895"},"schema_version":"1.0"},"canonical_sha256":"9c596aa4cee12d3cb475b3146d33c5350faa4378cd766283cfb0195a589f2a3c","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:35:07.807648Z","signature_b64":"e5Q0nXIbhBqVN5IhnOhtBoKxNjqFJnx8pG0npsdbYY621se3p5LSjKNqiSrJ39OC7MwrJuxWQ2P6e8aWYnhwBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9c596aa4cee12d3cb475b3146d33c5350faa4378cd766283cfb0195a589f2a3c","last_reissued_at":"2026-07-05T09:35:07.807105Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:35:07.807105Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2401.09417","source_version":3,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:35:07Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"IrHTtTxXQZzeRkMJecVjTE+q83l1ep4if5phJlJxx8M1wIIKmUcZjdO9PRqueyrxjqAOWG2kCKKNbYeiZIewDw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-03T17:12:32.743726Z"},"content_sha256":"f81cfadd409b84eed7d8e37833e9a13b22875e95ef47f70903fbbd8a8a9499df","schema_version":"1.0","event_id":"sha256:f81cfadd409b84eed7d8e37833e9a13b22875e95ef47f70903fbbd8a8a9499df"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:TRMWVJGO4EWTZNDVWMKG2M6FGU","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Vision Mamba: Efficient Visual Representation Learning with Bidirectional State Space Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"A vision backbone built on bidirectional Mamba blocks outperforms DeiT transformers in accuracy and efficiency on image tasks.","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Bencheng Liao, Lianghui Zhu, Qian Zhang, Wenyu Liu, Xinggang Wang, Xinlong Wang","submitted_at":"2024-01-17T18:56:18Z","abstract_excerpt":"Recently the state space models (SSMs) with efficient hardware-aware designs, i.e., the Mamba deep learning model, have shown great potential for long sequence modeling. Meanwhile building efficient and generic vision backbones purely upon SSMs is an appealing direction. However, representing visual data is challenging for SSMs due to the position-sensitivity of visual data and the requirement of global context for visual understanding. In this paper, we show that the reliance on self-attention for visual representation learning is not necessary and propose a new generic vision backbone with b"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"we show that the reliance on self-attention for visual representation learning is not necessary and propose a new generic vision backbone with bidirectional Mamba blocks (Vim)","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"That bidirectional state space models with position embeddings can sufficiently capture the position-sensitive nature and global context requirements of visual data without any self-attention mechanism.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"Vim is a bidirectional Mamba vision backbone that outperforms DeiT in accuracy on standard tasks while being substantially faster and more memory-efficient for high-resolution images.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"A vision backbone built on bidirectional Mamba blocks outperforms DeiT transformers in accuracy and efficiency on image tasks.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"4b5372af60a32de8186daee2d5e75a3fc450645de2b52076f7a2f1fbbe0ed187"},"source":{"id":"2401.09417","kind":"arxiv","version":3},"verdict":{"id":"8ba42227-bc7c-409c-8b04-2998ccbd5f19","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-11T21:29:46.195528Z","strongest_claim":"we show that the reliance on self-attention for visual representation learning is not necessary and propose a new generic vision backbone with bidirectional Mamba blocks (Vim)","one_line_summary":"Vim is a bidirectional Mamba vision backbone that outperforms DeiT in accuracy on standard tasks while being substantially faster and more memory-efficient for high-resolution images.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"That bidirectional state space models with position embeddings can sufficiently capture the position-sensitive nature and global context requirements of visual data without any self-attention mechanism.","pith_extraction_headline":"A vision backbone built on bidirectional Mamba blocks outperforms DeiT transformers in accuracy and efficiency on image tasks."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.09417/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":80,"sample":[{"doi":"","year":2022,"title":"Beit: BERT pre-training of image transformers","work_id":"d08eb839-564e-4390-8b31-b0f193edf14f","ref_index":1,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2023,"title":"2-d ssm: A general spatial layer for visual transformers","work_id":"bd15b6f6-80c6-4e4e-8a31-4a7c849ae02c","ref_index":2,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2023,"title":"Introducing our multimodal models","work_id":"7cc15674-e922-4c8b-9e4e-5892b9feab03","ref_index":3,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2019,"title":"Cai, Z. and Vasconcelos, N. Cascade r-cnn: High quality object detection and instance segmentation. TPAMI, 2019","work_id":"67b9136a-e90d-45f1-83ab-583018344a92","ref_index":4,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2021,"title":"Emerging properties in self-supervised vision transformers","work_id":"dea4a0f9-6c25-416f-a021-50ee1fd9fb1e","ref_index":5,"cited_arxiv_id":"","is_internal_anchor":false}],"resolved_work":80,"snapshot_sha256":"e5bf742ddd668e6dff053d70b9ce4c869364810299a4615ebec239bfe671c83d","internal_anchors":11},"formal_canon":{"evidence_count":2,"snapshot_sha256":"045286793b7c0ff19dddd2a3e9236a006d89c0ee851fa3c50c8144b0f331086b"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"8ba42227-bc7c-409c-8b04-2998ccbd5f19"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:35:07Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"DtYh3sBeEZskQMY8nANit4ZQY+vRGgi4zXY1uKl6Oz2eRCGnJ/keaAciMkVB5yhL1VF+3R4Jit2jLhaAAiEUAQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-03T17:12:32.744511Z"},"content_sha256":"2b1ed2b1e7f7286d7db1f1a2d4266f8652cb478b6359a701bfdd879230d6d87b","schema_version":"1.0","event_id":"sha256:2b1ed2b1e7f7286d7db1f1a2d4266f8652cb478b6359a701bfdd879230d6d87b"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/TRMWVJGO4EWTZNDVWMKG2M6FGU/bundle.json","state_url":"https://pith.science/pith/TRMWVJGO4EWTZNDVWMKG2M6FGU/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/TRMWVJGO4EWTZNDVWMKG2M6FGU/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-03T17:12:32Z","links":{"resolver":"https://pith.science/pith/TRMWVJGO4EWTZNDVWMKG2M6FGU","bundle":"https://pith.science/pith/TRMWVJGO4EWTZNDVWMKG2M6FGU/bundle.json","state":"https://pith.science/pith/TRMWVJGO4EWTZNDVWMKG2M6FGU/state.json","well_known_bundle":"https://pith.science/.well-known/pith/TRMWVJGO4EWTZNDVWMKG2M6FGU/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:TRMWVJGO4EWTZNDVWMKG2M6FGU","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"08893014a214076870d708961c8371b9aebd84ca543d8ef36511c9d76d677895","cross_cats_sorted":["cs.LG"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-01-17T18:56:18Z","title_canon_sha256":"8014b4e743e657cb6756a563aefd1361d584179b6474d0d0c0594fbe86770470"},"schema_version":"1.0","source":{"id":"2401.09417","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2401.09417","created_at":"2026-07-05T09:35:07Z"},{"alias_kind":"arxiv_version","alias_value":"2401.09417v3","created_at":"2026-07-05T09:35:07Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.09417","created_at":"2026-07-05T09:35:07Z"},{"alias_kind":"pith_short_12","alias_value":"TRMWVJGO4EWT","created_at":"2026-07-05T09:35:07Z"},{"alias_kind":"pith_short_16","alias_value":"TRMWVJGO4EWTZNDV","created_at":"2026-07-05T09:35:07Z"},{"alias_kind":"pith_short_8","alias_value":"TRMWVJGO","created_at":"2026-07-05T09:35:07Z"}],"graph_snapshots":[{"event_id":"sha256:2b1ed2b1e7f7286d7db1f1a2d4266f8652cb478b6359a701bfdd879230d6d87b","target":"graph","created_at":"2026-07-05T09:35:07Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"we show that the reliance on self-attention for visual representation learning is not necessary and propose a new generic vision backbone with bidirectional Mamba blocks (Vim)"},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"That bidirectional state space models with position embeddings can sufficiently capture the position-sensitive nature and global context requirements of visual data without any self-attention mechanism."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"Vim is a bidirectional Mamba vision backbone that outperforms DeiT in accuracy on standard tasks while being substantially faster and more memory-efficient for high-resolution images."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"A vision backbone built on bidirectional Mamba blocks outperforms DeiT transformers in accuracy and efficiency on image tasks."}],"snapshot_sha256":"4b5372af60a32de8186daee2d5e75a3fc450645de2b52076f7a2f1fbbe0ed187"},"formal_canon":{"evidence_count":2,"snapshot_sha256":"045286793b7c0ff19dddd2a3e9236a006d89c0ee851fa3c50c8144b0f331086b"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2401.09417/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Recently the state space models (SSMs) with efficient hardware-aware designs, i.e., the Mamba deep learning model, have shown great potential for long sequence modeling. Meanwhile building efficient and generic vision backbones purely upon SSMs is an appealing direction. However, representing visual data is challenging for SSMs due to the position-sensitivity of visual data and the requirement of global context for visual understanding. In this paper, we show that the reliance on self-attention for visual representation learning is not necessary and propose a new generic vision backbone with b","authors_text":"Bencheng Liao, Lianghui Zhu, Qian Zhang, Wenyu Liu, Xinggang Wang, Xinlong Wang","cross_cats":["cs.LG"],"headline":"A vision backbone built on bidirectional Mamba blocks outperforms DeiT transformers in accuracy and efficiency on image tasks.","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-01-17T18:56:18Z","title":"Vision Mamba: Efficient Visual Representation Learning with Bidirectional State Space Model"},"references":{"count":80,"internal_anchors":11,"resolved_work":80,"sample":[{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":1,"title":"Beit: BERT pre-training of image transformers","work_id":"d08eb839-564e-4390-8b31-b0f193edf14f","year":2022},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":2,"title":"2-d ssm: A general spatial layer for visual transformers","work_id":"bd15b6f6-80c6-4e4e-8a31-4a7c849ae02c","year":2023},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":3,"title":"Introducing our multimodal models","work_id":"7cc15674-e922-4c8b-9e4e-5892b9feab03","year":2023},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":4,"title":"Cai, Z. and Vasconcelos, N. Cascade r-cnn: High quality object detection and instance segmentation. TPAMI, 2019","work_id":"67b9136a-e90d-45f1-83ab-583018344a92","year":2019},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":5,"title":"Emerging properties in self-supervised vision transformers","work_id":"dea4a0f9-6c25-416f-a021-50ee1fd9fb1e","year":2021}],"snapshot_sha256":"e5bf742ddd668e6dff053d70b9ce4c869364810299a4615ebec239bfe671c83d"},"source":{"id":"2401.09417","kind":"arxiv","version":3},"verdict":{"created_at":"2026-05-11T21:29:46.195528Z","id":"8ba42227-bc7c-409c-8b04-2998ccbd5f19","model_set":{"reader":"grok-4.3"},"one_line_summary":"Vim is a bidirectional Mamba vision backbone that outperforms DeiT in accuracy on standard tasks while being substantially faster and more memory-efficient for high-resolution images.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"A vision backbone built on bidirectional Mamba blocks outperforms DeiT transformers in accuracy and efficiency on image tasks.","strongest_claim":"we show that the reliance on self-attention for visual representation learning is not necessary and propose a new generic vision backbone with bidirectional Mamba blocks (Vim)","weakest_assumption":"That bidirectional state space models with position embeddings can sufficiently capture the position-sensitive nature and global context requirements of visual data without any self-attention mechanism."}},"verdict_id":"8ba42227-bc7c-409c-8b04-2998ccbd5f19"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:f81cfadd409b84eed7d8e37833e9a13b22875e95ef47f70903fbbd8a8a9499df","target":"record","created_at":"2026-07-05T09:35:07Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"08893014a214076870d708961c8371b9aebd84ca543d8ef36511c9d76d677895","cross_cats_sorted":["cs.LG"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-01-17T18:56:18Z","title_canon_sha256":"8014b4e743e657cb6756a563aefd1361d584179b6474d0d0c0594fbe86770470"},"schema_version":"1.0","source":{"id":"2401.09417","kind":"arxiv","version":3}},"canonical_sha256":"9c596aa4cee12d3cb475b3146d33c5350faa4378cd766283cfb0195a589f2a3c","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"9c596aa4cee12d3cb475b3146d33c5350faa4378cd766283cfb0195a589f2a3c","first_computed_at":"2026-07-05T09:35:07.807105Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T09:35:07.807105Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"e5Q0nXIbhBqVN5IhnOhtBoKxNjqFJnx8pG0npsdbYY621se3p5LSjKNqiSrJ39OC7MwrJuxWQ2P6e8aWYnhwBw==","signature_status":"signed_v1","signed_at":"2026-07-05T09:35:07.807648Z","signed_message":"canonical_sha256_bytes"},"source_id":"2401.09417","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:f81cfadd409b84eed7d8e37833e9a13b22875e95ef47f70903fbbd8a8a9499df","sha256:2b1ed2b1e7f7286d7db1f1a2d4266f8652cb478b6359a701bfdd879230d6d87b"],"state_sha256":"df61f8991dcfeb7d3a7b30e0ab33d37aae6812b2342785e7fa960af232339a82"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"hSx01pBZyU1ZzW1hvE1yGSPKkfj6SPQCZtX7EcE+Mz2+U4Ma1dFwC6XJUcukbagSaJJkriLXnj+VrSOa/ATcBg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-03T17:12:32.750607Z","bundle_sha256":"542afd431479384b6726ff9b38085de7ddf6d21af385f8e2658adbb7345465c4"}}