{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:B2RJGFV3J7Q7PNELXY5MPM6GAG","short_pith_number":"pith:B2RJGFV3","schema_version":"1.0","canonical_sha256":"0ea29316bb4fe1f7b48bbe3ac7b3c60180c76a3816ee51ad85462281904da086","source":{"kind":"arxiv","id":"2506.02555","version":1},"attestation_state":"computed","paper":{"title":"SurgVLM: A Large Vision-Language Model and Systematic Evaluation Benchmark for Surgical Intelligence","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chang Han Low, Erli Zhang, Jiaan Zhang, Jian Jiang, Junde Wu, Qi Dou, Xiaochun Cao, Xiaojun Jia, Yang Liu, Yueming Jin, Yutong Ban, Yuxuan Wang, Zhitao Zeng, Zhu Zhuo, Zilong Zheng","submitted_at":"2025-06-03T07:44:41Z","abstract_excerpt":"Foundation models have achieved transformative success across biomedical domains by enabling holistic understanding of multimodal data. However, their application in surgery remains underexplored. Surgical intelligence presents unique challenges - requiring surgical visual perception, temporal analysis, and reasoning. Existing general-purpose vision-language models fail to address these needs due to insufficient domain-specific supervision and the lack of a large-scale high-quality surgical database. To bridge this gap, we propose SurgVLM, one of the first large vision-language foundation mode"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.02555","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2025-06-03T07:44:41Z","cross_cats_sorted":[],"title_canon_sha256":"d0b5ac9ab9c614e8d5500da9ec7ef806086e3457d772258936a1a521b2bea0a1","abstract_canon_sha256":"a1c4b526bc782809b303147a4cf137358df197c2ae329fdd6ed0b7392af01ec1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:15:04.048845Z","signature_b64":"iZE8iIO7s5Z9O08DbtyMKwlM9+BpbbL98C3W5u+BMLzsH+yrNTcKmI5uO4YiYEKdvAEFj8+Xa7Zw2ueNyFvsDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0ea29316bb4fe1f7b48bbe3ac7b3c60180c76a3816ee51ad85462281904da086","last_reissued_at":"2026-07-05T11:15:04.048322Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:15:04.048322Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SurgVLM: A Large Vision-Language Model and Systematic Evaluation Benchmark for Surgical Intelligence","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chang Han Low, Erli Zhang, Jiaan Zhang, Jian Jiang, Junde Wu, Qi Dou, Xiaochun Cao, Xiaojun Jia, Yang Liu, Yueming Jin, Yutong Ban, Yuxuan Wang, Zhitao Zeng, Zhu Zhuo, Zilong Zheng","submitted_at":"2025-06-03T07:44:41Z","abstract_excerpt":"Foundation models have achieved transformative success across biomedical domains by enabling holistic understanding of multimodal data. However, their application in surgery remains underexplored. Surgical intelligence presents unique challenges - requiring surgical visual perception, temporal analysis, and reasoning. Existing general-purpose vision-language models fail to address these needs due to insufficient domain-specific supervision and the lack of a large-scale high-quality surgical database. To bridge this gap, we propose SurgVLM, one of the first large vision-language foundation mode"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.02555","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.02555/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.02555","created_at":"2026-07-05T11:15:04.048388+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.02555v1","created_at":"2026-07-05T11:15:04.048388+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.02555","created_at":"2026-07-05T11:15:04.048388+00:00"},{"alias_kind":"pith_short_12","alias_value":"B2RJGFV3J7Q7","created_at":"2026-07-05T11:15:04.048388+00:00"},{"alias_kind":"pith_short_16","alias_value":"B2RJGFV3J7Q7PNEL","created_at":"2026-07-05T11:15:04.048388+00:00"},{"alias_kind":"pith_short_8","alias_value":"B2RJGFV3","created_at":"2026-07-05T11:15:04.048388+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25905","citing_title":"SurgAtlas: A Large-Scale Surgical Video-Language Dataset with 2,391 Hours of Open and Minimally Invasive Surgery","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07433","citing_title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","ref_index":268,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08071","citing_title":"SurgiQ: A Large-Scale Multi-Domain Benchmark for Evaluating Surgical Understanding in Large Language Models","ref_index":87,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20319","citing_title":"SurgCoT: Advancing Spatiotemporal Reasoning in Surgical Videos through a Chain-of-Thought Benchmark","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10233","citing_title":"Adapting 2D Multi-Modal Large Language Model for 3D CT Image Analysis","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09037","citing_title":"SiMing-Bench: Evaluating Procedural Correctness from Continuous Interactions in Clinical Skill Videos","ref_index":40,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/B2RJGFV3J7Q7PNELXY5MPM6GAG","json":"https://pith.science/pith/B2RJGFV3J7Q7PNELXY5MPM6GAG.json","graph_json":"https://pith.science/api/pith-number/B2RJGFV3J7Q7PNELXY5MPM6GAG/graph.json","events_json":"https://pith.science/api/pith-number/B2RJGFV3J7Q7PNELXY5MPM6GAG/events.json","paper":"https://pith.science/paper/B2RJGFV3"},"agent_actions":{"view_html":"https://pith.science/pith/B2RJGFV3J7Q7PNELXY5MPM6GAG","download_json":"https://pith.science/pith/B2RJGFV3J7Q7PNELXY5MPM6GAG.json","view_paper":"https://pith.science/paper/B2RJGFV3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.02555&json=true","fetch_graph":"https://pith.science/api/pith-number/B2RJGFV3J7Q7PNELXY5MPM6GAG/graph.json","fetch_events":"https://pith.science/api/pith-number/B2RJGFV3J7Q7PNELXY5MPM6GAG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/B2RJGFV3J7Q7PNELXY5MPM6GAG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/B2RJGFV3J7Q7PNELXY5MPM6GAG/action/storage_attestation","attest_author":"https://pith.science/pith/B2RJGFV3J7Q7PNELXY5MPM6GAG/action/author_attestation","sign_citation":"https://pith.science/pith/B2RJGFV3J7Q7PNELXY5MPM6GAG/action/citation_signature","submit_replication":"https://pith.science/pith/B2RJGFV3J7Q7PNELXY5MPM6GAG/action/replication_record"}},"created_at":"2026-07-05T11:15:04.048388+00:00","updated_at":"2026-07-05T11:15:04.048388+00:00"}