{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:RIDROOR47IDXCVGAEOU66UH2X5","short_pith_number":"pith:RIDROOR4","schema_version":"1.0","canonical_sha256":"8a07173a3cfa077154c023a9ef50fabf76017dd8ae5246dc1ea7ae91f19d4ac0","source":{"kind":"arxiv","id":"2501.11347","version":2},"attestation_state":"computed","paper":{"title":"EndoChat: Grounded Multimodal Large Language Model for Endoscopic Surgery","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Fan Zhang, Guankun Wang, Hongbin Liu, Hongliang Ren, Jiazheng Wang, Jinlin Wu, Junyi Wang, Kun Yuan, Long Bai, Nassir Navab, Nicolas Padoy, Tianxu Jiang, Xiting He, Zhen Chen, Zhen Lei, Zhen Li","submitted_at":"2025-01-20T09:12:06Z","abstract_excerpt":"Recently, Multimodal Large Language Models (MLLMs) have demonstrated their immense potential in computer-aided diagnosis and decision-making. In the context of robotic-assisted surgery, MLLMs can serve as effective tools for surgical training and guidance. However, there is still a lack of MLLMs specialized for surgical scene understanding in clinical applications. In this work, we introduce EndoChat to address various dialogue paradigms and subtasks in surgical scene understanding that surgeons encounter. To train our EndoChat, we construct the Surg-396K dataset through a novel pipeline that "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.11347","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-01-20T09:12:06Z","cross_cats_sorted":[],"title_canon_sha256":"f2e98232531bf4ef69569580436cbb6f6eee6e8cddd22db488cdbd5351e420c8","abstract_canon_sha256":"e9f0b1c0a25ac81412de99f9b95dc1addbc9d4c84c85d14825c4bb56e19f9ae9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:31:53.248608Z","signature_b64":"ZBD2zLigG3UrWv68dX38bIjGuf8JVCwcWaxdacJ07IP1UWKhobGi1HW69OIg/Bnk2G5Mub8AhjCGOaipDRSqCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8a07173a3cfa077154c023a9ef50fabf76017dd8ae5246dc1ea7ae91f19d4ac0","last_reissued_at":"2026-07-05T10:31:53.248102Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:31:53.248102Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"EndoChat: Grounded Multimodal Large Language Model for Endoscopic Surgery","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Fan Zhang, Guankun Wang, Hongbin Liu, Hongliang Ren, Jiazheng Wang, Jinlin Wu, Junyi Wang, Kun Yuan, Long Bai, Nassir Navab, Nicolas Padoy, Tianxu Jiang, Xiting He, Zhen Chen, Zhen Lei, Zhen Li","submitted_at":"2025-01-20T09:12:06Z","abstract_excerpt":"Recently, Multimodal Large Language Models (MLLMs) have demonstrated their immense potential in computer-aided diagnosis and decision-making. In the context of robotic-assisted surgery, MLLMs can serve as effective tools for surgical training and guidance. However, there is still a lack of MLLMs specialized for surgical scene understanding in clinical applications. In this work, we introduce EndoChat to address various dialogue paradigms and subtasks in surgical scene understanding that surgeons encounter. To train our EndoChat, we construct the Surg-396K dataset through a novel pipeline that "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.11347","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.11347/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.11347","created_at":"2026-07-05T10:31:53.248161+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.11347v2","created_at":"2026-07-05T10:31:53.248161+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.11347","created_at":"2026-07-05T10:31:53.248161+00:00"},{"alias_kind":"pith_short_12","alias_value":"RIDROOR47IDX","created_at":"2026-07-05T10:31:53.248161+00:00"},{"alias_kind":"pith_short_16","alias_value":"RIDROOR47IDXCVGA","created_at":"2026-07-05T10:31:53.248161+00:00"},{"alias_kind":"pith_short_8","alias_value":"RIDROOR4","created_at":"2026-07-05T10:31:53.248161+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.23068","citing_title":"RoboSurg-VQA: A Multimodal Benchmark for Surgical Segmentation-Aware Visual Question Answering","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11208","citing_title":"Hi-GaTA: Hierarchical Gated Temporal Aggregation Adapter for Surgical Video Report Generation","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2512.06581","citing_title":"MedGRPO: Multi-Task Reinforcement Learning for Heterogeneous Medical Video Understanding","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11208","citing_title":"Hi-GaTA: Hierarchical Gated Temporal Aggregation Adapter for Surgical Video Report Generation","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RIDROOR47IDXCVGAEOU66UH2X5","json":"https://pith.science/pith/RIDROOR47IDXCVGAEOU66UH2X5.json","graph_json":"https://pith.science/api/pith-number/RIDROOR47IDXCVGAEOU66UH2X5/graph.json","events_json":"https://pith.science/api/pith-number/RIDROOR47IDXCVGAEOU66UH2X5/events.json","paper":"https://pith.science/paper/RIDROOR4"},"agent_actions":{"view_html":"https://pith.science/pith/RIDROOR47IDXCVGAEOU66UH2X5","download_json":"https://pith.science/pith/RIDROOR47IDXCVGAEOU66UH2X5.json","view_paper":"https://pith.science/paper/RIDROOR4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.11347&json=true","fetch_graph":"https://pith.science/api/pith-number/RIDROOR47IDXCVGAEOU66UH2X5/graph.json","fetch_events":"https://pith.science/api/pith-number/RIDROOR47IDXCVGAEOU66UH2X5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RIDROOR47IDXCVGAEOU66UH2X5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RIDROOR47IDXCVGAEOU66UH2X5/action/storage_attestation","attest_author":"https://pith.science/pith/RIDROOR47IDXCVGAEOU66UH2X5/action/author_attestation","sign_citation":"https://pith.science/pith/RIDROOR47IDXCVGAEOU66UH2X5/action/citation_signature","submit_replication":"https://pith.science/pith/RIDROOR47IDXCVGAEOU66UH2X5/action/replication_record"}},"created_at":"2026-07-05T10:31:53.248161+00:00","updated_at":"2026-07-05T10:31:53.248161+00:00"}