{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:MQFAKTEIOHNDS4OYQ4ZTGNGJ2B","short_pith_number":"pith:MQFAKTEI","schema_version":"1.0","canonical_sha256":"640a054c8871da3971d887333334c9d063d3cdab5745856ddf225591ba08f97d","source":{"kind":"arxiv","id":"2509.00357","version":1},"attestation_state":"computed","paper":{"title":"SurgLLM: A Versatile Large Multimodal Model with Spatial Focus and Temporal Awareness for Surgical Video Understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Danny T.M. Chan, Hongbin Liu, Jiebo Luo, Jinlin Wu, Kun Yuan, Nassir Navab, Xingjian Luo, Zhen Chen, Zhen Lei","submitted_at":"2025-08-30T04:36:41Z","abstract_excerpt":"Surgical video understanding is crucial for facilitating Computer-Assisted Surgery (CAS) systems. Despite significant progress in existing studies, two major limitations persist, including inadequate visual content perception and insufficient temporal awareness in surgical videos, and hinder the development of versatile CAS solutions. In this work, we propose the SurgLLM framework, an effective large multimodal model tailored for versatile surgical video understanding tasks with enhanced spatial focus and temporal awareness. Specifically, to empower the spatial focus of surgical videos, we fir"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2509.00357","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-08-30T04:36:41Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"2929be83e2ed9fa2784164cd53124b67ba81d4fd7a1c6a6afb7ec0a1a4141a2c","abstract_canon_sha256":"39d56ca961a853c62e4d80777da496baa319279def773f6e0c0c4a809818f0c9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:02:14.540702Z","signature_b64":"8JmHBzSaEMFC5XoIXS+BrZ+6EpMdRKuPuLq5cYOpn5as+vbOxOIuUPVeuPFLQLHrTbWjlNXmgH17bnPQtKpVBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"640a054c8871da3971d887333334c9d063d3cdab5745856ddf225591ba08f97d","last_reissued_at":"2026-07-05T12:02:14.540147Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:02:14.540147Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SurgLLM: A Versatile Large Multimodal Model with Spatial Focus and Temporal Awareness for Surgical Video Understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Danny T.M. Chan, Hongbin Liu, Jiebo Luo, Jinlin Wu, Kun Yuan, Nassir Navab, Xingjian Luo, Zhen Chen, Zhen Lei","submitted_at":"2025-08-30T04:36:41Z","abstract_excerpt":"Surgical video understanding is crucial for facilitating Computer-Assisted Surgery (CAS) systems. Despite significant progress in existing studies, two major limitations persist, including inadequate visual content perception and insufficient temporal awareness in surgical videos, and hinder the development of versatile CAS solutions. In this work, we propose the SurgLLM framework, an effective large multimodal model tailored for versatile surgical video understanding tasks with enhanced spatial focus and temporal awareness. Specifically, to empower the spatial focus of surgical videos, we fir"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.00357","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.00357/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2509.00357","created_at":"2026-07-05T12:02:14.540208+00:00"},{"alias_kind":"arxiv_version","alias_value":"2509.00357v1","created_at":"2026-07-05T12:02:14.540208+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.00357","created_at":"2026-07-05T12:02:14.540208+00:00"},{"alias_kind":"pith_short_12","alias_value":"MQFAKTEIOHND","created_at":"2026-07-05T12:02:14.540208+00:00"},{"alias_kind":"pith_short_16","alias_value":"MQFAKTEIOHNDS4OY","created_at":"2026-07-05T12:02:14.540208+00:00"},{"alias_kind":"pith_short_8","alias_value":"MQFAKTEI","created_at":"2026-07-05T12:02:14.540208+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25905","citing_title":"SurgAtlas: A Large-Scale Surgical Video-Language Dataset with 2,391 Hours of Open and Minimally Invasive Surgery","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11740","citing_title":"UniReason-Med: A Shared Grounded Reasoning Interface for 2D-to-3D Transfer in Medical VQA","ref_index":248,"is_internal_anchor":false},{"citing_arxiv_id":"2512.06581","citing_title":"MedGRPO: Multi-Task Reinforcement Learning for Heterogeneous Medical Video Understanding","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20319","citing_title":"SurgCoT: Advancing Spatiotemporal Reasoning in Surgical Videos through a Chain-of-Thought Benchmark","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MQFAKTEIOHNDS4OYQ4ZTGNGJ2B","json":"https://pith.science/pith/MQFAKTEIOHNDS4OYQ4ZTGNGJ2B.json","graph_json":"https://pith.science/api/pith-number/MQFAKTEIOHNDS4OYQ4ZTGNGJ2B/graph.json","events_json":"https://pith.science/api/pith-number/MQFAKTEIOHNDS4OYQ4ZTGNGJ2B/events.json","paper":"https://pith.science/paper/MQFAKTEI"},"agent_actions":{"view_html":"https://pith.science/pith/MQFAKTEIOHNDS4OYQ4ZTGNGJ2B","download_json":"https://pith.science/pith/MQFAKTEIOHNDS4OYQ4ZTGNGJ2B.json","view_paper":"https://pith.science/paper/MQFAKTEI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2509.00357&json=true","fetch_graph":"https://pith.science/api/pith-number/MQFAKTEIOHNDS4OYQ4ZTGNGJ2B/graph.json","fetch_events":"https://pith.science/api/pith-number/MQFAKTEIOHNDS4OYQ4ZTGNGJ2B/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MQFAKTEIOHNDS4OYQ4ZTGNGJ2B/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MQFAKTEIOHNDS4OYQ4ZTGNGJ2B/action/storage_attestation","attest_author":"https://pith.science/pith/MQFAKTEIOHNDS4OYQ4ZTGNGJ2B/action/author_attestation","sign_citation":"https://pith.science/pith/MQFAKTEIOHNDS4OYQ4ZTGNGJ2B/action/citation_signature","submit_replication":"https://pith.science/pith/MQFAKTEIOHNDS4OYQ4ZTGNGJ2B/action/replication_record"}},"created_at":"2026-07-05T12:02:14.540208+00:00","updated_at":"2026-07-05T12:02:14.540208+00:00"}