{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:JEPNEK465VUXGQPQI7XGHKDNNP","short_pith_number":"pith:JEPNEK46","canonical_record":{"source":{"id":"2405.20421","kind":"arxiv","version":5},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-05-30T18:56:01Z","cross_cats_sorted":[],"title_canon_sha256":"dc291432b6d3599f26eb30f1ca8a32bd5c35191ac58315a40413a75f5b853f0b","abstract_canon_sha256":"0a5a7f4667fb9be4b4b036b15c641ed0e940cc595062b7f524d7e64594713989"},"schema_version":"1.0"},"canonical_sha256":"491ed22b9eed697341f047ee63a86d6be4babac3d0410b09e2644925bc62b3bb","source":{"kind":"arxiv","id":"2405.20421","version":5},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2405.20421","created_at":"2026-07-05T11:19:19Z"},{"alias_kind":"arxiv_version","alias_value":"2405.20421v5","created_at":"2026-07-05T11:19:19Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.20421","created_at":"2026-07-05T11:19:19Z"},{"alias_kind":"pith_short_12","alias_value":"JEPNEK465VUX","created_at":"2026-07-05T11:19:19Z"},{"alias_kind":"pith_short_16","alias_value":"JEPNEK465VUXGQPQ","created_at":"2026-07-05T11:19:19Z"},{"alias_kind":"pith_short_8","alias_value":"JEPNEK46","created_at":"2026-07-05T11:19:19Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:JEPNEK465VUXGQPQI7XGHKDNNP","target":"record","payload":{"canonical_record":{"source":{"id":"2405.20421","kind":"arxiv","version":5},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-05-30T18:56:01Z","cross_cats_sorted":[],"title_canon_sha256":"dc291432b6d3599f26eb30f1ca8a32bd5c35191ac58315a40413a75f5b853f0b","abstract_canon_sha256":"0a5a7f4667fb9be4b4b036b15c641ed0e940cc595062b7f524d7e64594713989"},"schema_version":"1.0"},"canonical_sha256":"491ed22b9eed697341f047ee63a86d6be4babac3d0410b09e2644925bc62b3bb","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:19:19.525746Z","signature_b64":"NoAJ2MqAxoa4WKHKYziHrJ4g4gI/iAyf8qGB7oKgd6wC1ceo9kmdrBcoWiTnKUt3ZUrvAoLblFLMSGTwjAIpAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"491ed22b9eed697341f047ee63a86d6be4babac3d0410b09e2644925bc62b3bb","last_reissued_at":"2026-07-05T11:19:19.525240Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:19:19.525240Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2405.20421","source_version":5,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:19:19Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"lZppU+I1lCrmwJK44Dma2ZhIFoRimioMsB9T1fOEaZTpPF3eqfjYEiT0zUigUqLerIsBv74Rcqf4c9Dzmy0UDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T23:37:15.827263Z"},"content_sha256":"bbb00a002e0e0ca529f203ecf4a0a21488b8c21f5a431870ef9717a20e93f95c","schema_version":"1.0","event_id":"sha256:bbb00a002e0e0ca529f203ecf4a0a21488b8c21f5a431870ef9717a20e93f95c"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:JEPNEK465VUXGQPQI7XGHKDNNP","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Worse than Random? An Embarrassingly Simple Probing Evaluation of Large Multimodal Models in Medical VQA","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Qianqi Yan, Xiang Yue, Xin Eric Wang, Xuehai He","submitted_at":"2024-05-30T18:56:01Z","abstract_excerpt":"Large Multimodal Models (LMMs) have shown remarkable progress in medical Visual Question Answering (Med-VQA), achieving high accuracy on existing benchmarks. However, their reliability under robust evaluation is questionable. This study reveals that when subjected to simple probing evaluation, state-of-the-art models perform worse than random guessing on medical diagnosis questions. To address this critical evaluation problem, we introduce the Probing Evaluation for Medical Diagnosis (ProbMed) dataset to rigorously assess LMM performance in medical imaging through probing evaluation and proced"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.20421","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.20421/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:19:19Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"gFba2V0z1x/PCYmwtFYIcvQdLuOUIoQqZC+jO+7fM3o2N0zG1lCW0liqin1cH+TNLVLbmhBqS7nCw5qKWX3VBg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T23:37:15.827865Z"},"content_sha256":"2d50f10e89080790cccef6925293c13521e651c57a802bb3a48312fc97ef352e","schema_version":"1.0","event_id":"sha256:2d50f10e89080790cccef6925293c13521e651c57a802bb3a48312fc97ef352e"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/JEPNEK465VUXGQPQI7XGHKDNNP/bundle.json","state_url":"https://pith.science/pith/JEPNEK465VUXGQPQI7XGHKDNNP/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/JEPNEK465VUXGQPQI7XGHKDNNP/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-04T23:37:15Z","links":{"resolver":"https://pith.science/pith/JEPNEK465VUXGQPQI7XGHKDNNP","bundle":"https://pith.science/pith/JEPNEK465VUXGQPQI7XGHKDNNP/bundle.json","state":"https://pith.science/pith/JEPNEK465VUXGQPQI7XGHKDNNP/state.json","well_known_bundle":"https://pith.science/.well-known/pith/JEPNEK465VUXGQPQI7XGHKDNNP/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:JEPNEK465VUXGQPQI7XGHKDNNP","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"0a5a7f4667fb9be4b4b036b15c641ed0e940cc595062b7f524d7e64594713989","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-05-30T18:56:01Z","title_canon_sha256":"dc291432b6d3599f26eb30f1ca8a32bd5c35191ac58315a40413a75f5b853f0b"},"schema_version":"1.0","source":{"id":"2405.20421","kind":"arxiv","version":5}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2405.20421","created_at":"2026-07-05T11:19:19Z"},{"alias_kind":"arxiv_version","alias_value":"2405.20421v5","created_at":"2026-07-05T11:19:19Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.20421","created_at":"2026-07-05T11:19:19Z"},{"alias_kind":"pith_short_12","alias_value":"JEPNEK465VUX","created_at":"2026-07-05T11:19:19Z"},{"alias_kind":"pith_short_16","alias_value":"JEPNEK465VUXGQPQ","created_at":"2026-07-05T11:19:19Z"},{"alias_kind":"pith_short_8","alias_value":"JEPNEK46","created_at":"2026-07-05T11:19:19Z"}],"graph_snapshots":[{"event_id":"sha256:2d50f10e89080790cccef6925293c13521e651c57a802bb3a48312fc97ef352e","target":"graph","created_at":"2026-07-05T11:19:19Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2405.20421/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Large Multimodal Models (LMMs) have shown remarkable progress in medical Visual Question Answering (Med-VQA), achieving high accuracy on existing benchmarks. However, their reliability under robust evaluation is questionable. This study reveals that when subjected to simple probing evaluation, state-of-the-art models perform worse than random guessing on medical diagnosis questions. To address this critical evaluation problem, we introduce the Probing Evaluation for Medical Diagnosis (ProbMed) dataset to rigorously assess LMM performance in medical imaging through probing evaluation and proced","authors_text":"Qianqi Yan, Xiang Yue, Xin Eric Wang, Xuehai He","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-05-30T18:56:01Z","title":"Worse than Random? An Embarrassingly Simple Probing Evaluation of Large Multimodal Models in Medical VQA"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.20421","kind":"arxiv","version":5},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:bbb00a002e0e0ca529f203ecf4a0a21488b8c21f5a431870ef9717a20e93f95c","target":"record","created_at":"2026-07-05T11:19:19Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"0a5a7f4667fb9be4b4b036b15c641ed0e940cc595062b7f524d7e64594713989","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-05-30T18:56:01Z","title_canon_sha256":"dc291432b6d3599f26eb30f1ca8a32bd5c35191ac58315a40413a75f5b853f0b"},"schema_version":"1.0","source":{"id":"2405.20421","kind":"arxiv","version":5}},"canonical_sha256":"491ed22b9eed697341f047ee63a86d6be4babac3d0410b09e2644925bc62b3bb","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"491ed22b9eed697341f047ee63a86d6be4babac3d0410b09e2644925bc62b3bb","first_computed_at":"2026-07-05T11:19:19.525240Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:19:19.525240Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"NoAJ2MqAxoa4WKHKYziHrJ4g4gI/iAyf8qGB7oKgd6wC1ceo9kmdrBcoWiTnKUt3ZUrvAoLblFLMSGTwjAIpAQ==","signature_status":"signed_v1","signed_at":"2026-07-05T11:19:19.525746Z","signed_message":"canonical_sha256_bytes"},"source_id":"2405.20421","source_kind":"arxiv","source_version":5}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:bbb00a002e0e0ca529f203ecf4a0a21488b8c21f5a431870ef9717a20e93f95c","sha256:2d50f10e89080790cccef6925293c13521e651c57a802bb3a48312fc97ef352e"],"state_sha256":"d91a8346f0f30177b1ef1c4f28d644d14d90d72c6ada232b25e70857aeda966f"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"AfYcpHxglS0kRK5QKWlO6CR7RojNW7kxUTmn1UZeP6axzVpLSN8XUbG1pkoUdSk/ILk4w7XngDQmgzBmUVDQBA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-04T23:37:15.835868Z","bundle_sha256":"c8e1e3e143139d9e9d7913b09670a686e4f38e6a86605eea9c89f4e191016f0f"}}