{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:4ZAZQKUJ6CEAX5BCW75UE3GO6E","short_pith_number":"pith:4ZAZQKUJ","schema_version":"1.0","canonical_sha256":"e641982a89f0880bf422b7fb426ccef11a785a08a31aa8416f087e1d49ba7b3a","source":{"kind":"arxiv","id":"2406.19369","version":1},"attestation_state":"computed","paper":{"title":"Mamba or RWKV: Exploring High-Quality and High-Efficiency Segment Anything Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chen Change Loy, Haobo Yuan, Lu Qi, Ming-Hsuan Yang, Shuicheng Yan, Tao Zhang, Xiangtai Li","submitted_at":"2024-06-27T17:49:25Z","abstract_excerpt":"Transformer-based segmentation methods face the challenge of efficient inference when dealing with high-resolution images. Recently, several linear attention architectures, such as Mamba and RWKV, have attracted much attention as they can process long sequences efficiently. In this work, we focus on designing an efficient segment-anything model by exploring these different architectures. Specifically, we design a mixed backbone that contains convolution and RWKV operation, which achieves the best for both accuracy and efficiency. In addition, we design an efficient decoder to utilize the multi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.19369","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-06-27T17:49:25Z","cross_cats_sorted":[],"title_canon_sha256":"45d2e78efeb4903cd36a5b09f2fa1f3ed8a3c4ac1487052fe50d2405e238300d","abstract_canon_sha256":"9e504be2e84dff600b993108c17f5780672eaf9f1a2270cbaed7203dbc9ceb7a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:37:33.662972Z","signature_b64":"CJGQX6wcuUKBO0TVAGSIIgi4K5QEBxYn6FB/u4q0lcYvqWUmIB+9NYCvJsYKXYBmjbpIRFcXzGbtyniZMXILDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e641982a89f0880bf422b7fb426ccef11a785a08a31aa8416f087e1d49ba7b3a","last_reissued_at":"2026-07-05T08:37:33.662493Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:37:33.662493Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Mamba or RWKV: Exploring High-Quality and High-Efficiency Segment Anything Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chen Change Loy, Haobo Yuan, Lu Qi, Ming-Hsuan Yang, Shuicheng Yan, Tao Zhang, Xiangtai Li","submitted_at":"2024-06-27T17:49:25Z","abstract_excerpt":"Transformer-based segmentation methods face the challenge of efficient inference when dealing with high-resolution images. Recently, several linear attention architectures, such as Mamba and RWKV, have attracted much attention as they can process long sequences efficiently. In this work, we focus on designing an efficient segment-anything model by exploring these different architectures. Specifically, we design a mixed backbone that contains convolution and RWKV operation, which achieves the best for both accuracy and efficiency. In addition, we design an efficient decoder to utilize the multi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.19369","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.19369/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.19369","created_at":"2026-07-05T08:37:33.662547+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.19369v1","created_at":"2026-07-05T08:37:33.662547+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.19369","created_at":"2026-07-05T08:37:33.662547+00:00"},{"alias_kind":"pith_short_12","alias_value":"4ZAZQKUJ6CEA","created_at":"2026-07-05T08:37:33.662547+00:00"},{"alias_kind":"pith_short_16","alias_value":"4ZAZQKUJ6CEAX5BC","created_at":"2026-07-05T08:37:33.662547+00:00"},{"alias_kind":"pith_short_8","alias_value":"4ZAZQKUJ","created_at":"2026-07-05T08:37:33.662547+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.10395","citing_title":"Efficient RWKV-based Representation Learning for 3D Point Clouds","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2410.04960","citing_title":"On Efficient Variants of Segment Anything Model: A Survey","ref_index":136,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17278","citing_title":"PestVL-Net: Enabling Multimodal Pest Learning via Fine-grained Vision-Language Interaction","ref_index":51,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4ZAZQKUJ6CEAX5BCW75UE3GO6E","json":"https://pith.science/pith/4ZAZQKUJ6CEAX5BCW75UE3GO6E.json","graph_json":"https://pith.science/api/pith-number/4ZAZQKUJ6CEAX5BCW75UE3GO6E/graph.json","events_json":"https://pith.science/api/pith-number/4ZAZQKUJ6CEAX5BCW75UE3GO6E/events.json","paper":"https://pith.science/paper/4ZAZQKUJ"},"agent_actions":{"view_html":"https://pith.science/pith/4ZAZQKUJ6CEAX5BCW75UE3GO6E","download_json":"https://pith.science/pith/4ZAZQKUJ6CEAX5BCW75UE3GO6E.json","view_paper":"https://pith.science/paper/4ZAZQKUJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.19369&json=true","fetch_graph":"https://pith.science/api/pith-number/4ZAZQKUJ6CEAX5BCW75UE3GO6E/graph.json","fetch_events":"https://pith.science/api/pith-number/4ZAZQKUJ6CEAX5BCW75UE3GO6E/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4ZAZQKUJ6CEAX5BCW75UE3GO6E/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4ZAZQKUJ6CEAX5BCW75UE3GO6E/action/storage_attestation","attest_author":"https://pith.science/pith/4ZAZQKUJ6CEAX5BCW75UE3GO6E/action/author_attestation","sign_citation":"https://pith.science/pith/4ZAZQKUJ6CEAX5BCW75UE3GO6E/action/citation_signature","submit_replication":"https://pith.science/pith/4ZAZQKUJ6CEAX5BCW75UE3GO6E/action/replication_record"}},"created_at":"2026-07-05T08:37:33.662547+00:00","updated_at":"2026-07-05T08:37:33.662547+00:00"}