{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:IJSYMLYF2GDQ3YK3XCSGCI3PRH","short_pith_number":"pith:IJSYMLYF","schema_version":"1.0","canonical_sha256":"4265862f05d1870de15bb8a461236f89eced635bf23b5ca496c0533d114484fb","source":{"kind":"arxiv","id":"2410.09750","version":1},"attestation_state":"computed","paper":{"title":"Surgical-LLaVA: Toward Surgical Scenario Understanding via Large Language and Vision Models","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Chang Wook Jeong, Juseong Jin","submitted_at":"2024-10-13T07:12:35Z","abstract_excerpt":"Conversation agents powered by large language models are revolutionizing the way we interact with visual data. Recently, large vision-language models (LVLMs) have been extensively studied for both images and videos. However, these studies typically focus on common scenarios. In this work, we introduce an LVLM specifically designed for surgical scenarios. We integrate visual representations of surgical images and videos into the language feature space. Consequently, we establish a LVLM model, Surgical-LLaVA, fine-tuned on instruction following data of surgical scenarios. Our experiments demonst"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.09750","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-10-13T07:12:35Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"05c22adab0090ce1693b81ef7b3cc5d103124b3fe36ed9de555fc0c1e962edcd","abstract_canon_sha256":"268a34bec92f2b5bb7bd4d74da360284ae0cfb19db2317d3f3ce6f065740a7e4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:19:43.566586Z","signature_b64":"lre0F5g1K2NECEsxhX0nT++C40P8FhKKfbVNFGf0jkCUpTxNrx3mcII2gFa7wN71haR7CnAeq9wHuSJTYgFzAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4265862f05d1870de15bb8a461236f89eced635bf23b5ca496c0533d114484fb","last_reissued_at":"2026-07-05T09:19:43.566084Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:19:43.566084Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Surgical-LLaVA: Toward Surgical Scenario Understanding via Large Language and Vision Models","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Chang Wook Jeong, Juseong Jin","submitted_at":"2024-10-13T07:12:35Z","abstract_excerpt":"Conversation agents powered by large language models are revolutionizing the way we interact with visual data. Recently, large vision-language models (LVLMs) have been extensively studied for both images and videos. However, these studies typically focus on common scenarios. In this work, we introduce an LVLM specifically designed for surgical scenarios. We integrate visual representations of surgical images and videos into the language feature space. Consequently, we establish a LVLM model, Surgical-LLaVA, fine-tuned on instruction following data of surgical scenarios. Our experiments demonst"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.09750","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.09750/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.09750","created_at":"2026-07-05T09:19:43.566145+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.09750v1","created_at":"2026-07-05T09:19:43.566145+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.09750","created_at":"2026-07-05T09:19:43.566145+00:00"},{"alias_kind":"pith_short_12","alias_value":"IJSYMLYF2GDQ","created_at":"2026-07-05T09:19:43.566145+00:00"},{"alias_kind":"pith_short_16","alias_value":"IJSYMLYF2GDQ3YK3","created_at":"2026-07-05T09:19:43.566145+00:00"},{"alias_kind":"pith_short_8","alias_value":"IJSYMLYF","created_at":"2026-07-05T09:19:43.566145+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07433","citing_title":"Watch, Remember, Reason: Human-View Video Understanding with MLLMs","ref_index":266,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27484","citing_title":"Fine-tuning a multimodal large language model for clinician-grade autism behavioral scoring from short home videos","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08712","citing_title":"From Articulated Kinematics to Routed Visual Control for Action-Conditioned Surgical Video Generation","ref_index":31,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IJSYMLYF2GDQ3YK3XCSGCI3PRH","json":"https://pith.science/pith/IJSYMLYF2GDQ3YK3XCSGCI3PRH.json","graph_json":"https://pith.science/api/pith-number/IJSYMLYF2GDQ3YK3XCSGCI3PRH/graph.json","events_json":"https://pith.science/api/pith-number/IJSYMLYF2GDQ3YK3XCSGCI3PRH/events.json","paper":"https://pith.science/paper/IJSYMLYF"},"agent_actions":{"view_html":"https://pith.science/pith/IJSYMLYF2GDQ3YK3XCSGCI3PRH","download_json":"https://pith.science/pith/IJSYMLYF2GDQ3YK3XCSGCI3PRH.json","view_paper":"https://pith.science/paper/IJSYMLYF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.09750&json=true","fetch_graph":"https://pith.science/api/pith-number/IJSYMLYF2GDQ3YK3XCSGCI3PRH/graph.json","fetch_events":"https://pith.science/api/pith-number/IJSYMLYF2GDQ3YK3XCSGCI3PRH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IJSYMLYF2GDQ3YK3XCSGCI3PRH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IJSYMLYF2GDQ3YK3XCSGCI3PRH/action/storage_attestation","attest_author":"https://pith.science/pith/IJSYMLYF2GDQ3YK3XCSGCI3PRH/action/author_attestation","sign_citation":"https://pith.science/pith/IJSYMLYF2GDQ3YK3XCSGCI3PRH/action/citation_signature","submit_replication":"https://pith.science/pith/IJSYMLYF2GDQ3YK3XCSGCI3PRH/action/replication_record"}},"created_at":"2026-07-05T09:19:43.566145+00:00","updated_at":"2026-07-05T09:19:43.566145+00:00"}