{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:VKGQYLDXGS3DFPFHEE7XVGXDUO","short_pith_number":"pith:VKGQYLDX","schema_version":"1.0","canonical_sha256":"aa8d0c2c7734b632bca7213f7a9ae3a3bf213b4c45ed7d26dd143f1cad2f83f1","source":{"kind":"arxiv","id":"2405.13949","version":1},"attestation_state":"computed","paper":{"title":"PitVQA: Image-grounded Text Embedding LLM for Visual Question Answering in Pituitary Surgery","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Adrito Das, Danail Stoyanov, Danyal Z. Khan, Hani J. Marcus, Matthew J. Clarkson, Mengya Xu, Mobarakol Islam, Runlong He, Sophia Bano","submitted_at":"2024-05-22T19:30:24Z","abstract_excerpt":"Visual Question Answering (VQA) within the surgical domain, utilizing Large Language Models (LLMs), offers a distinct opportunity to improve intra-operative decision-making and facilitate intuitive surgeon-AI interaction. However, the development of LLMs for surgical VQA is hindered by the scarcity of diverse and extensive datasets with complex reasoning tasks. Moreover, contextual fusion of the image and text modalities remains an open research challenge due to the inherent differences between these two types of information and the complexity involved in aligning them. This paper introduces P"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.13949","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-05-22T19:30:24Z","cross_cats_sorted":[],"title_canon_sha256":"35d0179afc63789b865ef78862c69dd98097aa2e5eefbf862bcceb5b504db996","abstract_canon_sha256":"3b47b9de1d6a3587fd7fe3adcd988e65043deace6dfed114e3e348284e9c5a2b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:22:08.587691Z","signature_b64":"P8qKTAWos3QcECHmPoGv92zgsy8Ildz76LvbJ2DvWdNutMvDkCruXuX/Dkpuvgq7GY+/Pcuo0VZd0TsHjNhTCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"aa8d0c2c7734b632bca7213f7a9ae3a3bf213b4c45ed7d26dd143f1cad2f83f1","last_reissued_at":"2026-07-05T08:22:08.587283Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:22:08.587283Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"PitVQA: Image-grounded Text Embedding LLM for Visual Question Answering in Pituitary Surgery","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Adrito Das, Danail Stoyanov, Danyal Z. Khan, Hani J. Marcus, Matthew J. Clarkson, Mengya Xu, Mobarakol Islam, Runlong He, Sophia Bano","submitted_at":"2024-05-22T19:30:24Z","abstract_excerpt":"Visual Question Answering (VQA) within the surgical domain, utilizing Large Language Models (LLMs), offers a distinct opportunity to improve intra-operative decision-making and facilitate intuitive surgeon-AI interaction. However, the development of LLMs for surgical VQA is hindered by the scarcity of diverse and extensive datasets with complex reasoning tasks. Moreover, contextual fusion of the image and text modalities remains an open research challenge due to the inherent differences between these two types of information and the complexity involved in aligning them. This paper introduces P"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.13949","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.13949/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.13949","created_at":"2026-07-05T08:22:08.587340+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.13949v1","created_at":"2026-07-05T08:22:08.587340+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.13949","created_at":"2026-07-05T08:22:08.587340+00:00"},{"alias_kind":"pith_short_12","alias_value":"VKGQYLDXGS3D","created_at":"2026-07-05T08:22:08.587340+00:00"},{"alias_kind":"pith_short_16","alias_value":"VKGQYLDXGS3DFPFH","created_at":"2026-07-05T08:22:08.587340+00:00"},{"alias_kind":"pith_short_8","alias_value":"VKGQYLDX","created_at":"2026-07-05T08:22:08.587340+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VKGQYLDXGS3DFPFHEE7XVGXDUO","json":"https://pith.science/pith/VKGQYLDXGS3DFPFHEE7XVGXDUO.json","graph_json":"https://pith.science/api/pith-number/VKGQYLDXGS3DFPFHEE7XVGXDUO/graph.json","events_json":"https://pith.science/api/pith-number/VKGQYLDXGS3DFPFHEE7XVGXDUO/events.json","paper":"https://pith.science/paper/VKGQYLDX"},"agent_actions":{"view_html":"https://pith.science/pith/VKGQYLDXGS3DFPFHEE7XVGXDUO","download_json":"https://pith.science/pith/VKGQYLDXGS3DFPFHEE7XVGXDUO.json","view_paper":"https://pith.science/paper/VKGQYLDX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.13949&json=true","fetch_graph":"https://pith.science/api/pith-number/VKGQYLDXGS3DFPFHEE7XVGXDUO/graph.json","fetch_events":"https://pith.science/api/pith-number/VKGQYLDXGS3DFPFHEE7XVGXDUO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VKGQYLDXGS3DFPFHEE7XVGXDUO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VKGQYLDXGS3DFPFHEE7XVGXDUO/action/storage_attestation","attest_author":"https://pith.science/pith/VKGQYLDXGS3DFPFHEE7XVGXDUO/action/author_attestation","sign_citation":"https://pith.science/pith/VKGQYLDXGS3DFPFHEE7XVGXDUO/action/citation_signature","submit_replication":"https://pith.science/pith/VKGQYLDXGS3DFPFHEE7XVGXDUO/action/replication_record"}},"created_at":"2026-07-05T08:22:08.587340+00:00","updated_at":"2026-07-05T08:22:08.587340+00:00"}