{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:Y2NJVV4HXY4FZWGNKDDW2FKMGI","short_pith_number":"pith:Y2NJVV4H","schema_version":"1.0","canonical_sha256":"c69a9ad787be385cd8cd50c76d154c321e7fa67fd8254777e9635f8fe2821fbf","source":{"kind":"arxiv","id":"2302.13069","version":1},"attestation_state":"computed","paper":{"title":"Medical visual question answering using joint self-supervised learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Jing Mei, Tanveer Syeda-Mahmood, Yiqin Yu, Yuan Zhou","submitted_at":"2023-02-25T12:12:22Z","abstract_excerpt":"Visual Question Answering (VQA) becomes one of the most active research problems in the medical imaging domain. A well-known VQA challenge is the intrinsic diversity between the image and text modalities, and in the medical VQA task, there is another critical problem relying on the limited size of labelled image-question-answer data. In this study we propose an encoder-decoder framework that leverages the image-text joint representation learned from large-scaled medical image-caption data and adapted to the small-sized medical VQA task. The encoder embeds across the image-text dual modalities "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2302.13069","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-02-25T12:12:22Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"50a285992cbb3ab32667af0473e72df6387d48bc2545f0101f4d40e24e4af96f","abstract_canon_sha256":"a89cc1a32b804603bae21204ba1adbf16100f5afd876a166cac461a4f4e2f23c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:45:42.920446Z","signature_b64":"f6zv+lU7bYCdoXoJDei1XtTgPIFGb4tdc9VTx/DavGvU6mgwQSSPn6y59pMkC19BfRPJSRtMcctcey9dYF/nBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c69a9ad787be385cd8cd50c76d154c321e7fa67fd8254777e9635f8fe2821fbf","last_reissued_at":"2026-07-05T05:45:42.920075Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:45:42.920075Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Medical visual question answering using joint self-supervised learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Jing Mei, Tanveer Syeda-Mahmood, Yiqin Yu, Yuan Zhou","submitted_at":"2023-02-25T12:12:22Z","abstract_excerpt":"Visual Question Answering (VQA) becomes one of the most active research problems in the medical imaging domain. A well-known VQA challenge is the intrinsic diversity between the image and text modalities, and in the medical VQA task, there is another critical problem relying on the limited size of labelled image-question-answer data. In this study we propose an encoder-decoder framework that leverages the image-text joint representation learned from large-scaled medical image-caption data and adapted to the small-sized medical VQA task. The encoder embeds across the image-text dual modalities "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2302.13069","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2302.13069/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2302.13069","created_at":"2026-07-05T05:45:42.920123+00:00"},{"alias_kind":"arxiv_version","alias_value":"2302.13069v1","created_at":"2026-07-05T05:45:42.920123+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2302.13069","created_at":"2026-07-05T05:45:42.920123+00:00"},{"alias_kind":"pith_short_12","alias_value":"Y2NJVV4HXY4F","created_at":"2026-07-05T05:45:42.920123+00:00"},{"alias_kind":"pith_short_16","alias_value":"Y2NJVV4HXY4FZWGN","created_at":"2026-07-05T05:45:42.920123+00:00"},{"alias_kind":"pith_short_8","alias_value":"Y2NJVV4H","created_at":"2026-07-05T05:45:42.920123+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.19319","citing_title":"MedVQA-TREE: A Multimodal Reasoning and Retrieval Framework for Sarcopenia Prediction","ref_index":54,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Y2NJVV4HXY4FZWGNKDDW2FKMGI","json":"https://pith.science/pith/Y2NJVV4HXY4FZWGNKDDW2FKMGI.json","graph_json":"https://pith.science/api/pith-number/Y2NJVV4HXY4FZWGNKDDW2FKMGI/graph.json","events_json":"https://pith.science/api/pith-number/Y2NJVV4HXY4FZWGNKDDW2FKMGI/events.json","paper":"https://pith.science/paper/Y2NJVV4H"},"agent_actions":{"view_html":"https://pith.science/pith/Y2NJVV4HXY4FZWGNKDDW2FKMGI","download_json":"https://pith.science/pith/Y2NJVV4HXY4FZWGNKDDW2FKMGI.json","view_paper":"https://pith.science/paper/Y2NJVV4H","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2302.13069&json=true","fetch_graph":"https://pith.science/api/pith-number/Y2NJVV4HXY4FZWGNKDDW2FKMGI/graph.json","fetch_events":"https://pith.science/api/pith-number/Y2NJVV4HXY4FZWGNKDDW2FKMGI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Y2NJVV4HXY4FZWGNKDDW2FKMGI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Y2NJVV4HXY4FZWGNKDDW2FKMGI/action/storage_attestation","attest_author":"https://pith.science/pith/Y2NJVV4HXY4FZWGNKDDW2FKMGI/action/author_attestation","sign_citation":"https://pith.science/pith/Y2NJVV4HXY4FZWGNKDDW2FKMGI/action/citation_signature","submit_replication":"https://pith.science/pith/Y2NJVV4HXY4FZWGNKDDW2FKMGI/action/replication_record"}},"created_at":"2026-07-05T05:45:42.920123+00:00","updated_at":"2026-07-05T05:45:42.920123+00:00"}