{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:GUZH6BLKAYSRLLNG3LPKNXOCMC","short_pith_number":"pith:GUZH6BLK","schema_version":"1.0","canonical_sha256":"35327f056a062515ada6dadea6ddc26081f44da2e9426295a8a1fbddb5f86a4a","source":{"kind":"arxiv","id":"2311.11969","version":1},"attestation_state":"computed","paper":{"title":"SA-Med2D-20M Dataset: Segment Anything in 2D Medical Imaging with 20 Million masks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"eess.IV","authors_text":"Haoyu Wang, Hui Sun, Jianpin Chen, Jilong Chen, Jin Ye, Junjun He, Junlong Cheng, Lei Jiang, Min Zhu, Shaoting Zhang, Tianbin Li, Yanzhou Su, Yu Qiao, Zhongying Deng, Ziyan Huang","submitted_at":"2023-11-20T17:59:03Z","abstract_excerpt":"Segment Anything Model (SAM) has achieved impressive results for natural image segmentation with input prompts such as points and bounding boxes. Its success largely owes to massive labeled training data. However, directly applying SAM to medical image segmentation cannot perform well because SAM lacks medical knowledge -- it does not use medical images for training. To incorporate medical knowledge into SAM, we introduce SA-Med2D-20M, a large-scale segmentation dataset of 2D medical images built upon numerous public and private datasets. It consists of 4.6 million 2D medical images and 19.7 m"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.11969","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"eess.IV","submitted_at":"2023-11-20T17:59:03Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"ffaa1754a78fa9c5d813df0ccbd809516361faf295160049420ab031cd025d1f","abstract_canon_sha256":"ac7c9c1b1038c8a401e96d0d3e6e29186a0e860ce912e156217f0eff19dd4483"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:14:41.557207Z","signature_b64":"kKwZnxYboX1B0W2vtECWWzAfvlac9UauCpakLxctfSF+JF164Y4yOXIY94pPL6BUkLTRxduSjKf2xSn5DSl5Bw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"35327f056a062515ada6dadea6ddc26081f44da2e9426295a8a1fbddb5f86a4a","last_reissued_at":"2026-07-05T07:14:41.556746Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:14:41.556746Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SA-Med2D-20M Dataset: Segment Anything in 2D Medical Imaging with 20 Million masks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"eess.IV","authors_text":"Haoyu Wang, Hui Sun, Jianpin Chen, Jilong Chen, Jin Ye, Junjun He, Junlong Cheng, Lei Jiang, Min Zhu, Shaoting Zhang, Tianbin Li, Yanzhou Su, Yu Qiao, Zhongying Deng, Ziyan Huang","submitted_at":"2023-11-20T17:59:03Z","abstract_excerpt":"Segment Anything Model (SAM) has achieved impressive results for natural image segmentation with input prompts such as points and bounding boxes. Its success largely owes to massive labeled training data. However, directly applying SAM to medical image segmentation cannot perform well because SAM lacks medical knowledge -- it does not use medical images for training. To incorporate medical knowledge into SAM, we introduce SA-Med2D-20M, a large-scale segmentation dataset of 2D medical images built upon numerous public and private datasets. It consists of 4.6 million 2D medical images and 19.7 m"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.11969","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.11969/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.11969","created_at":"2026-07-05T07:14:41.556828+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.11969v1","created_at":"2026-07-05T07:14:41.556828+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.11969","created_at":"2026-07-05T07:14:41.556828+00:00"},{"alias_kind":"pith_short_12","alias_value":"GUZH6BLKAYSR","created_at":"2026-07-05T07:14:41.556828+00:00"},{"alias_kind":"pith_short_16","alias_value":"GUZH6BLKAYSRLLNG","created_at":"2026-07-05T07:14:41.556828+00:00"},{"alias_kind":"pith_short_8","alias_value":"GUZH6BLK","created_at":"2026-07-05T07:14:41.556828+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.06760","citing_title":"MedSIGHT: Towards Grounded Visual Comprehension in Medical Large Vision-Language Models","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24492","citing_title":"Med-R2: An Adversarial Benchmark for Evidence-Grounded Reasoning in Medical VLMs","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10242","citing_title":"MedVeriSeg: Teaching MLLM-Based Medical Segmentation Models to Verify Query Validity Without Extra Training","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14805","citing_title":"From Boundaries to Semantics: Prompt-Guided Multi-Task Learning for Petrographic Thin-section Segmentation","ref_index":37,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GUZH6BLKAYSRLLNG3LPKNXOCMC","json":"https://pith.science/pith/GUZH6BLKAYSRLLNG3LPKNXOCMC.json","graph_json":"https://pith.science/api/pith-number/GUZH6BLKAYSRLLNG3LPKNXOCMC/graph.json","events_json":"https://pith.science/api/pith-number/GUZH6BLKAYSRLLNG3LPKNXOCMC/events.json","paper":"https://pith.science/paper/GUZH6BLK"},"agent_actions":{"view_html":"https://pith.science/pith/GUZH6BLKAYSRLLNG3LPKNXOCMC","download_json":"https://pith.science/pith/GUZH6BLKAYSRLLNG3LPKNXOCMC.json","view_paper":"https://pith.science/paper/GUZH6BLK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.11969&json=true","fetch_graph":"https://pith.science/api/pith-number/GUZH6BLKAYSRLLNG3LPKNXOCMC/graph.json","fetch_events":"https://pith.science/api/pith-number/GUZH6BLKAYSRLLNG3LPKNXOCMC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GUZH6BLKAYSRLLNG3LPKNXOCMC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GUZH6BLKAYSRLLNG3LPKNXOCMC/action/storage_attestation","attest_author":"https://pith.science/pith/GUZH6BLKAYSRLLNG3LPKNXOCMC/action/author_attestation","sign_citation":"https://pith.science/pith/GUZH6BLKAYSRLLNG3LPKNXOCMC/action/citation_signature","submit_replication":"https://pith.science/pith/GUZH6BLKAYSRLLNG3LPKNXOCMC/action/replication_record"}},"created_at":"2026-07-05T07:14:41.556828+00:00","updated_at":"2026-07-05T07:14:41.556828+00:00"}