{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:RFYZR7V3E2FKMKOKQBUWYDL6NZ","short_pith_number":"pith:RFYZR7V3","schema_version":"1.0","canonical_sha256":"897198febb268aa629ca80696c0d7e6e6faea4c2a0520cf8bb04497bbe2667ce","source":{"kind":"arxiv","id":"2306.12156","version":1},"attestation_state":"computed","paper":{"title":"Fast Segment Anything","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Jinqiao Wang, Ming Tang, Min Li, Tao Yu, Wenchao Ding, Xu Zhao, Yinglong Du, Yongqi An","submitted_at":"2023-06-21T10:08:29Z","abstract_excerpt":"The recently proposed segment anything model (SAM) has made a significant influence in many computer vision tasks. It is becoming a foundation step for many high-level tasks, like image segmentation, image caption, and image editing. However, its huge computation costs prevent it from wider applications in industry scenarios. The computation mainly comes from the Transformer architecture at high-resolution inputs. In this paper, we propose a speed-up alternative method for this fundamental task with comparable performance. By reformulating the task as segments-generation and prompting, we find"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.12156","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-06-21T10:08:29Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"8e340e5bdf295f9ecf80ea4953ce24ded0b3d3be769afeb58c769706c8def5fa","abstract_canon_sha256":"717e5a884269736e4e79f221bdd5359d654dd1202a4dcf0383181f805f1e879a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:23:11.223501Z","signature_b64":"CuzuhZJMXRow3C4mUoldAgxZqHgqY1PQ+qL2aYreoM0a4GN0ze2YbXNvks9KHS9QRknIWEtiSL4u/TqwR2APBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"897198febb268aa629ca80696c0d7e6e6faea4c2a0520cf8bb04497bbe2667ce","last_reissued_at":"2026-07-05T06:23:11.223085Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:23:11.223085Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Fast Segment Anything","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Jinqiao Wang, Ming Tang, Min Li, Tao Yu, Wenchao Ding, Xu Zhao, Yinglong Du, Yongqi An","submitted_at":"2023-06-21T10:08:29Z","abstract_excerpt":"The recently proposed segment anything model (SAM) has made a significant influence in many computer vision tasks. It is becoming a foundation step for many high-level tasks, like image segmentation, image caption, and image editing. However, its huge computation costs prevent it from wider applications in industry scenarios. The computation mainly comes from the Transformer architecture at high-resolution inputs. In this paper, we propose a speed-up alternative method for this fundamental task with comparable performance. By reformulating the task as segments-generation and prompting, we find"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.12156","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.12156/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.12156","created_at":"2026-07-05T06:23:11.223136+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.12156v1","created_at":"2026-07-05T06:23:11.223136+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.12156","created_at":"2026-07-05T06:23:11.223136+00:00"},{"alias_kind":"pith_short_12","alias_value":"RFYZR7V3E2FK","created_at":"2026-07-05T06:23:11.223136+00:00"},{"alias_kind":"pith_short_16","alias_value":"RFYZR7V3E2FKMKOK","created_at":"2026-07-05T06:23:11.223136+00:00"},{"alias_kind":"pith_short_8","alias_value":"RFYZR7V3","created_at":"2026-07-05T06:23:11.223136+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":38,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24649","citing_title":"Agentic Collaborative Cognition for Zero-Shot 3D Understanding","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2606.24649","citing_title":"Agentic Collaborative Cognition for Zero-Shot 3D Understanding","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23157","citing_title":"Bridging Semantics and Kinematics: A Modular Framework for Zero-Shot Robotic Manipulation","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22756","citing_title":"HERCULES: An Open-Source Simulation Framework for Heterogeneous Multi-Robot SLAM, Collaborative Perception, and Exploration","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20032","citing_title":"ReA-OVCD: Reliability-Aware Open-Vocabulary Change Detection via Semantic and Spatial Refinement","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07953","citing_title":"Unification of Closed-Open Industrial Detection Scenarios: New Large-Scale Benchmarks,Challenges and Baselines","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00124","citing_title":"Segmenting, Fast and Slow: Real-Time Open-Vocabulary Video Instance Segmentation with Dual-Path Processing","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06312","citing_title":"Meridian: Metric-Semantic Primitive Matching for Cross-View Geo-Localization Beyond Urban Environments","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31533","citing_title":"MV-GEL: Language-Driven Multi-View Geometric Entity Localization on Meshes","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30809","citing_title":"GaussLite: Online Task-Conditioned 3D Gaussian Splatting for Real-Time Robotic Mapping","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31157","citing_title":"Rethinking Foundation Model Collaboration: Enhancing Specialized Models through Proxy Task Reasoning","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26102","citing_title":"InstructSAM: Segment Any Instance with Any Instructions","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25495","citing_title":"RepSAM: Bridging Foundation Models to Robotic Vision via Representation-Guided Adaptation","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26862","citing_title":"RoadGIE: Towards A Global-Scale Aerial Benchmark for Generalizable Interactive Road Extraction","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26637","citing_title":"Enabling Extensible Embodied Capabilities with Tools","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28442","citing_title":"Self-Supervised Online Robot-Agnostic Traversability Estimation for Open-World Environments","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29505","citing_title":"ESAM++: Efficient Online 3D Perception on the Edge","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22439","citing_title":"Curvature-aware 3D length estimation of greenhouse cucumbers using RGB-D imaging and cubic spline arc-length integration","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2410.04960","citing_title":"On Efficient Variants of Segment Anything Model: A Survey","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17633","citing_title":"SparseSAM: Structured Sparsification of Activations in Segment Anything Models","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18013","citing_title":"TinySAM 2: Extreme Memory Compression for Efficient Track Anything Model","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19634","citing_title":"P2DNav: Panorama-to-Downview Reasoning for Zero-shot Vision-and-Language Navigation","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2509.19579","citing_title":"Terra: Hierarchical Terrain-Aware 3D Scene Graph for Task-Agnostic Outdoor Mapping","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2509.25699","citing_title":"AIM-CoT: Active Information-driven Multimodal Chain-of-Thought for Vision-Language Reasoning","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2306.14289","citing_title":"Faster Segment Anything: Towards Lightweight SAM for Mobile Applications","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RFYZR7V3E2FKMKOKQBUWYDL6NZ","json":"https://pith.science/pith/RFYZR7V3E2FKMKOKQBUWYDL6NZ.json","graph_json":"https://pith.science/api/pith-number/RFYZR7V3E2FKMKOKQBUWYDL6NZ/graph.json","events_json":"https://pith.science/api/pith-number/RFYZR7V3E2FKMKOKQBUWYDL6NZ/events.json","paper":"https://pith.science/paper/RFYZR7V3"},"agent_actions":{"view_html":"https://pith.science/pith/RFYZR7V3E2FKMKOKQBUWYDL6NZ","download_json":"https://pith.science/pith/RFYZR7V3E2FKMKOKQBUWYDL6NZ.json","view_paper":"https://pith.science/paper/RFYZR7V3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.12156&json=true","fetch_graph":"https://pith.science/api/pith-number/RFYZR7V3E2FKMKOKQBUWYDL6NZ/graph.json","fetch_events":"https://pith.science/api/pith-number/RFYZR7V3E2FKMKOKQBUWYDL6NZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RFYZR7V3E2FKMKOKQBUWYDL6NZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RFYZR7V3E2FKMKOKQBUWYDL6NZ/action/storage_attestation","attest_author":"https://pith.science/pith/RFYZR7V3E2FKMKOKQBUWYDL6NZ/action/author_attestation","sign_citation":"https://pith.science/pith/RFYZR7V3E2FKMKOKQBUWYDL6NZ/action/citation_signature","submit_replication":"https://pith.science/pith/RFYZR7V3E2FKMKOKQBUWYDL6NZ/action/replication_record"}},"created_at":"2026-07-05T06:23:11.223136+00:00","updated_at":"2026-07-05T06:23:11.223136+00:00"}