{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:UIJNBKVRUJKAXU36XK4NMI2JGW","short_pith_number":"pith:UIJNBKVR","schema_version":"1.0","canonical_sha256":"a212d0aab1a2540bd37ebab8d62349359764eb8fbcf71611e43b773f8f40a4e3","source":{"kind":"arxiv","id":"2104.01394","version":1},"attestation_state":"computed","paper":{"title":"MMBERT: Multimodal BERT Pretraining for Improved Medical VQA","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Adithi Devi, CV Jawahar, Minesh Mathew, U Deva Priyakumar, Viraj Bagal, Yash Khare","submitted_at":"2021-04-03T13:01:19Z","abstract_excerpt":"Images in the medical domain are fundamentally different from the general domain images. Consequently, it is infeasible to directly employ general domain Visual Question Answering (VQA) models for the medical domain. Additionally, medical images annotation is a costly and time-consuming process. To overcome these limitations, we propose a solution inspired by self-supervised pretraining of Transformer-style architectures for NLP, Vision and Language tasks. Our method involves learning richer medical image and text semantic representations using Masked Language Modeling (MLM) with image feature"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2104.01394","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2021-04-03T13:01:19Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"fe8e59a889eb098c2f18b25960967d8f0006e3b6665fcb9814bce094ba5f1e73","abstract_canon_sha256":"9d0b50a2d77a77c2a72f54230378bc2476837225b18c09bf30e9bd91a9fbe3bd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:28:59.557903Z","signature_b64":"jG2FgEc4fPtWrNvpLYLcWHl3yGeYznvbszdz9VGqUBcTAtE/OyOoqQL11CTvTRMSNKYhqOJblN/v2wercMQxBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a212d0aab1a2540bd37ebab8d62349359764eb8fbcf71611e43b773f8f40a4e3","last_reissued_at":"2026-07-05T02:28:59.557492Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:28:59.557492Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MMBERT: Multimodal BERT Pretraining for Improved Medical VQA","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Adithi Devi, CV Jawahar, Minesh Mathew, U Deva Priyakumar, Viraj Bagal, Yash Khare","submitted_at":"2021-04-03T13:01:19Z","abstract_excerpt":"Images in the medical domain are fundamentally different from the general domain images. Consequently, it is infeasible to directly employ general domain Visual Question Answering (VQA) models for the medical domain. Additionally, medical images annotation is a costly and time-consuming process. To overcome these limitations, we propose a solution inspired by self-supervised pretraining of Transformer-style architectures for NLP, Vision and Language tasks. Our method involves learning richer medical image and text semantic representations using Masked Language Modeling (MLM) with image feature"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2104.01394","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2104.01394/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2104.01394","created_at":"2026-07-05T02:28:59.557550+00:00"},{"alias_kind":"arxiv_version","alias_value":"2104.01394v1","created_at":"2026-07-05T02:28:59.557550+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2104.01394","created_at":"2026-07-05T02:28:59.557550+00:00"},{"alias_kind":"pith_short_12","alias_value":"UIJNBKVRUJKA","created_at":"2026-07-05T02:28:59.557550+00:00"},{"alias_kind":"pith_short_16","alias_value":"UIJNBKVRUJKAXU36","created_at":"2026-07-05T02:28:59.557550+00:00"},{"alias_kind":"pith_short_8","alias_value":"UIJNBKVR","created_at":"2026-07-05T02:28:59.557550+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UIJNBKVRUJKAXU36XK4NMI2JGW","json":"https://pith.science/pith/UIJNBKVRUJKAXU36XK4NMI2JGW.json","graph_json":"https://pith.science/api/pith-number/UIJNBKVRUJKAXU36XK4NMI2JGW/graph.json","events_json":"https://pith.science/api/pith-number/UIJNBKVRUJKAXU36XK4NMI2JGW/events.json","paper":"https://pith.science/paper/UIJNBKVR"},"agent_actions":{"view_html":"https://pith.science/pith/UIJNBKVRUJKAXU36XK4NMI2JGW","download_json":"https://pith.science/pith/UIJNBKVRUJKAXU36XK4NMI2JGW.json","view_paper":"https://pith.science/paper/UIJNBKVR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2104.01394&json=true","fetch_graph":"https://pith.science/api/pith-number/UIJNBKVRUJKAXU36XK4NMI2JGW/graph.json","fetch_events":"https://pith.science/api/pith-number/UIJNBKVRUJKAXU36XK4NMI2JGW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UIJNBKVRUJKAXU36XK4NMI2JGW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UIJNBKVRUJKAXU36XK4NMI2JGW/action/storage_attestation","attest_author":"https://pith.science/pith/UIJNBKVRUJKAXU36XK4NMI2JGW/action/author_attestation","sign_citation":"https://pith.science/pith/UIJNBKVRUJKAXU36XK4NMI2JGW/action/citation_signature","submit_replication":"https://pith.science/pith/UIJNBKVRUJKAXU36XK4NMI2JGW/action/replication_record"}},"created_at":"2026-07-05T02:28:59.557550+00:00","updated_at":"2026-07-05T02:28:59.557550+00:00"}