{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:5TH4SYX3YF753W7ZEHWKSOCMV2","short_pith_number":"pith:5TH4SYX3","schema_version":"1.0","canonical_sha256":"eccfc962fbc17fdddbf921eca9384caea5a2e6ad2475c89911ff003b4bfe2515","source":{"kind":"arxiv","id":"2404.00226","version":3},"attestation_state":"computed","paper":{"title":"Design as Desired: Utilizing Visual Question Answering for Multimodal Pre-training","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Baoliang Zhao, Faqin Lv, Haibo Jin, Hao Chen, Jun Li, Qiong Wang, Tongkun Su, Xi Zhang, Yin Hu","submitted_at":"2024-03-30T02:56:54Z","abstract_excerpt":"Multimodal pre-training demonstrates its potential in the medical domain, which learns medical visual representations from paired medical reports. However, many pre-training tasks require extra annotations from clinicians, and most of them fail to explicitly guide the model to learn the desired features of different pathologies. In this paper, we utilize Visual Question Answering (VQA) for multimodal pre-training to guide the framework focusing on targeted pathological features. We leverage descriptions in medical reports to design multi-granular question-answer pairs associated with different"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.00226","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-03-30T02:56:54Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"cd7aa2caa4620fbc083e2eed26e8ad7b2aa67085e0b06dcb0d9461840c3c06b5","abstract_canon_sha256":"044e26ffe2a47bc760462e9fc4a24ff31a82106a1662fc4c6f2de1af6745cb50"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:13:51.354304Z","signature_b64":"bdkh3ceyov/Qbs1j0tPVU6hXRZ5On4cS31ar30rYId9jNC5byt5EpSqKYifgPkdICL7etWZXyo+Ew/948OavDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"eccfc962fbc17fdddbf921eca9384caea5a2e6ad2475c89911ff003b4bfe2515","last_reissued_at":"2026-07-05T09:13:51.353852Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:13:51.353852Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Design as Desired: Utilizing Visual Question Answering for Multimodal Pre-training","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Baoliang Zhao, Faqin Lv, Haibo Jin, Hao Chen, Jun Li, Qiong Wang, Tongkun Su, Xi Zhang, Yin Hu","submitted_at":"2024-03-30T02:56:54Z","abstract_excerpt":"Multimodal pre-training demonstrates its potential in the medical domain, which learns medical visual representations from paired medical reports. However, many pre-training tasks require extra annotations from clinicians, and most of them fail to explicitly guide the model to learn the desired features of different pathologies. In this paper, we utilize Visual Question Answering (VQA) for multimodal pre-training to guide the framework focusing on targeted pathological features. We leverage descriptions in medical reports to design multi-granular question-answer pairs associated with different"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.00226","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.00226/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.00226","created_at":"2026-07-05T09:13:51.353907+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.00226v3","created_at":"2026-07-05T09:13:51.353907+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.00226","created_at":"2026-07-05T09:13:51.353907+00:00"},{"alias_kind":"pith_short_12","alias_value":"5TH4SYX3YF75","created_at":"2026-07-05T09:13:51.353907+00:00"},{"alias_kind":"pith_short_16","alias_value":"5TH4SYX3YF753W7Z","created_at":"2026-07-05T09:13:51.353907+00:00"},{"alias_kind":"pith_short_8","alias_value":"5TH4SYX3","created_at":"2026-07-05T09:13:51.353907+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5TH4SYX3YF753W7ZEHWKSOCMV2","json":"https://pith.science/pith/5TH4SYX3YF753W7ZEHWKSOCMV2.json","graph_json":"https://pith.science/api/pith-number/5TH4SYX3YF753W7ZEHWKSOCMV2/graph.json","events_json":"https://pith.science/api/pith-number/5TH4SYX3YF753W7ZEHWKSOCMV2/events.json","paper":"https://pith.science/paper/5TH4SYX3"},"agent_actions":{"view_html":"https://pith.science/pith/5TH4SYX3YF753W7ZEHWKSOCMV2","download_json":"https://pith.science/pith/5TH4SYX3YF753W7ZEHWKSOCMV2.json","view_paper":"https://pith.science/paper/5TH4SYX3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.00226&json=true","fetch_graph":"https://pith.science/api/pith-number/5TH4SYX3YF753W7ZEHWKSOCMV2/graph.json","fetch_events":"https://pith.science/api/pith-number/5TH4SYX3YF753W7ZEHWKSOCMV2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5TH4SYX3YF753W7ZEHWKSOCMV2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5TH4SYX3YF753W7ZEHWKSOCMV2/action/storage_attestation","attest_author":"https://pith.science/pith/5TH4SYX3YF753W7ZEHWKSOCMV2/action/author_attestation","sign_citation":"https://pith.science/pith/5TH4SYX3YF753W7ZEHWKSOCMV2/action/citation_signature","submit_replication":"https://pith.science/pith/5TH4SYX3YF753W7ZEHWKSOCMV2/action/replication_record"}},"created_at":"2026-07-05T09:13:51.353907+00:00","updated_at":"2026-07-05T09:13:51.353907+00:00"}