{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:ULKFVHMVCMKJKU3WYI67Q2H3LH","short_pith_number":"pith:ULKFVHMV","schema_version":"1.0","canonical_sha256":"a2d45a9d951314955376c23df868fb59c828685a51bce2bf4cb32ee65b3b9d89","source":{"kind":"arxiv","id":"2312.09579","version":1},"attestation_state":"computed","paper":{"title":"MobileSAMv2: Faster Segment Anything to Everything","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Chaoning Zhang, Choong Seon Hong, Dongshen Han, Jinwoo Choi, Sheng Zheng, Tae-Ho Kim","submitted_at":"2023-12-15T07:21:12Z","abstract_excerpt":"Segment anything model (SAM) addresses two practical yet challenging segmentation tasks: \\textbf{segment anything (SegAny)}, which utilizes a certain point to predict the mask for a single object of interest, and \\textbf{segment everything (SegEvery)}, which predicts the masks for all objects on the image. What makes SegAny slow for SAM is its heavyweight image encoder, which has been addressed by MobileSAM via decoupled knowledge distillation. The efficiency bottleneck of SegEvery with SAM, however, lies in its mask decoder because it needs to first generate numerous masks with redundant grid"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.09579","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2023-12-15T07:21:12Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"3bb73a07b405ce918759f8556237c8663d761a962d0c5ee17e0727e331704297","abstract_canon_sha256":"ae6f512e45c48fd1ce00fc0234114a66b1c3088df97a1ccc75350b952867c7df"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:24:27.547034Z","signature_b64":"zC901GbsVYt0gFx2qCCoGvMkmMabi2NHrGaRgJBmgOWKQEMrGTUTS81tfltUWuEV8srTIATPs9ThgdMFGp/FAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a2d45a9d951314955376c23df868fb59c828685a51bce2bf4cb32ee65b3b9d89","last_reissued_at":"2026-07-05T07:24:27.546500Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:24:27.546500Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MobileSAMv2: Faster Segment Anything to Everything","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Chaoning Zhang, Choong Seon Hong, Dongshen Han, Jinwoo Choi, Sheng Zheng, Tae-Ho Kim","submitted_at":"2023-12-15T07:21:12Z","abstract_excerpt":"Segment anything model (SAM) addresses two practical yet challenging segmentation tasks: \\textbf{segment anything (SegAny)}, which utilizes a certain point to predict the mask for a single object of interest, and \\textbf{segment everything (SegEvery)}, which predicts the masks for all objects on the image. What makes SegAny slow for SAM is its heavyweight image encoder, which has been addressed by MobileSAM via decoupled knowledge distillation. The efficiency bottleneck of SegEvery with SAM, however, lies in its mask decoder because it needs to first generate numerous masks with redundant grid"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.09579","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.09579/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.09579","created_at":"2026-07-05T07:24:27.546563+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.09579v1","created_at":"2026-07-05T07:24:27.546563+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.09579","created_at":"2026-07-05T07:24:27.546563+00:00"},{"alias_kind":"pith_short_12","alias_value":"ULKFVHMVCMKJ","created_at":"2026-07-05T07:24:27.546563+00:00"},{"alias_kind":"pith_short_16","alias_value":"ULKFVHMVCMKJKU3W","created_at":"2026-07-05T07:24:27.546563+00:00"},{"alias_kind":"pith_short_8","alias_value":"ULKFVHMV","created_at":"2026-07-05T07:24:27.546563+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25324","citing_title":"Efficient Remote Sensing Instance Segmentation with Linear-Time State Space Distilled Visual Foundation Models","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01860","citing_title":"DL-SLAM: Enabling High-Fidelity Gaussian Splatting SLAM in Dynamic Environments based on Dual-Level Probability","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08655","citing_title":"PhysGraph: A Physics-aware 3D Scene Graph for Perception and Reasoning","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00124","citing_title":"Segmenting, Fast and Slow: Real-Time Open-Vocabulary Video Instance Segmentation with Dual-Path Processing","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2410.04960","citing_title":"On Efficient Variants of Segment Anything Model: A Survey","ref_index":155,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07674","citing_title":"Weight Group-wise Post-Training Quantization for Medical Foundation Model","ref_index":28,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ULKFVHMVCMKJKU3WYI67Q2H3LH","json":"https://pith.science/pith/ULKFVHMVCMKJKU3WYI67Q2H3LH.json","graph_json":"https://pith.science/api/pith-number/ULKFVHMVCMKJKU3WYI67Q2H3LH/graph.json","events_json":"https://pith.science/api/pith-number/ULKFVHMVCMKJKU3WYI67Q2H3LH/events.json","paper":"https://pith.science/paper/ULKFVHMV"},"agent_actions":{"view_html":"https://pith.science/pith/ULKFVHMVCMKJKU3WYI67Q2H3LH","download_json":"https://pith.science/pith/ULKFVHMVCMKJKU3WYI67Q2H3LH.json","view_paper":"https://pith.science/paper/ULKFVHMV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.09579&json=true","fetch_graph":"https://pith.science/api/pith-number/ULKFVHMVCMKJKU3WYI67Q2H3LH/graph.json","fetch_events":"https://pith.science/api/pith-number/ULKFVHMVCMKJKU3WYI67Q2H3LH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ULKFVHMVCMKJKU3WYI67Q2H3LH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ULKFVHMVCMKJKU3WYI67Q2H3LH/action/storage_attestation","attest_author":"https://pith.science/pith/ULKFVHMVCMKJKU3WYI67Q2H3LH/action/author_attestation","sign_citation":"https://pith.science/pith/ULKFVHMVCMKJKU3WYI67Q2H3LH/action/citation_signature","submit_replication":"https://pith.science/pith/ULKFVHMVCMKJKU3WYI67Q2H3LH/action/replication_record"}},"created_at":"2026-07-05T07:24:27.546563+00:00","updated_at":"2026-07-05T07:24:27.546563+00:00"}