{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:WMMK4LKYHRFVFGEJO6QNTOERVV","short_pith_number":"pith:WMMK4LKY","schema_version":"1.0","canonical_sha256":"b318ae2d583c4b52988977a0d9b891ad4c0668f056ef4241676b18da993b333f","source":{"kind":"arxiv","id":"2311.07536","version":3},"attestation_state":"computed","paper":{"title":"A Comprehensive Evaluation of GPT-4V on Knowledge-Intensive Visual Question Answering","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Baotian Hu, Chenyang Lyu, Longyue Wang, Min Zhang, Wanqi Zhong, Wei Wang, Xinyu Chen, Yunxin Li","submitted_at":"2023-11-13T18:22:32Z","abstract_excerpt":"The emergence of multimodal large models (MLMs) has significantly advanced the field of visual understanding, offering remarkable capabilities in the realm of visual question answering (VQA). Yet, the true challenge lies in the domain of knowledge-intensive VQA tasks, which necessitate not just recognition of visual elements, but also a deep comprehension of the visual information in conjunction with a vast repository of learned knowledge. To uncover such capabilities of MLMs, particularly the newly introduced GPT-4V and Gemini, we provide an in-depth evaluation from three perspectives: 1) Com"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.07536","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2023-11-13T18:22:32Z","cross_cats_sorted":[],"title_canon_sha256":"99d1f79879a34d835b6d03bd630b33bbeb2fd6c48bf8385af6418c4448f48493","abstract_canon_sha256":"c862ee9e2784e17b78a3544ea87767cb7cded366674338498d130f33d31972e0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:58:41.210089Z","signature_b64":"4LhDksc+iMKOGadmO6wQO48umIpDQx7wCKWU4uEZRTAF89L2Ur9RZSycdzgFzxCWxC2iIMRA5w2KMww7sSc8AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b318ae2d583c4b52988977a0d9b891ad4c0668f056ef4241676b18da993b333f","last_reissued_at":"2026-07-05T08:58:41.209595Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:58:41.209595Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Comprehensive Evaluation of GPT-4V on Knowledge-Intensive Visual Question Answering","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Baotian Hu, Chenyang Lyu, Longyue Wang, Min Zhang, Wanqi Zhong, Wei Wang, Xinyu Chen, Yunxin Li","submitted_at":"2023-11-13T18:22:32Z","abstract_excerpt":"The emergence of multimodal large models (MLMs) has significantly advanced the field of visual understanding, offering remarkable capabilities in the realm of visual question answering (VQA). Yet, the true challenge lies in the domain of knowledge-intensive VQA tasks, which necessitate not just recognition of visual elements, but also a deep comprehension of the visual information in conjunction with a vast repository of learned knowledge. To uncover such capabilities of MLMs, particularly the newly introduced GPT-4V and Gemini, we provide an in-depth evaluation from three perspectives: 1) Com"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.07536","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.07536/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.07536","created_at":"2026-07-05T08:58:41.209657+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.07536v3","created_at":"2026-07-05T08:58:41.209657+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.07536","created_at":"2026-07-05T08:58:41.209657+00:00"},{"alias_kind":"pith_short_12","alias_value":"WMMK4LKYHRFV","created_at":"2026-07-05T08:58:41.209657+00:00"},{"alias_kind":"pith_short_16","alias_value":"WMMK4LKYHRFVFGEJ","created_at":"2026-07-05T08:58:41.209657+00:00"},{"alias_kind":"pith_short_8","alias_value":"WMMK4LKY","created_at":"2026-07-05T08:58:41.209657+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2506.20670","citing_title":"MMSearch-R1: Incentivizing LMMs to Search","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02934","citing_title":"PolyReal: A Benchmark for Real-World Polymer Science Workflows","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19567","citing_title":"Multi-modal Reasoning with LLMs for Visual Semantic Arithmetic","ref_index":11,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WMMK4LKYHRFVFGEJO6QNTOERVV","json":"https://pith.science/pith/WMMK4LKYHRFVFGEJO6QNTOERVV.json","graph_json":"https://pith.science/api/pith-number/WMMK4LKYHRFVFGEJO6QNTOERVV/graph.json","events_json":"https://pith.science/api/pith-number/WMMK4LKYHRFVFGEJO6QNTOERVV/events.json","paper":"https://pith.science/paper/WMMK4LKY"},"agent_actions":{"view_html":"https://pith.science/pith/WMMK4LKYHRFVFGEJO6QNTOERVV","download_json":"https://pith.science/pith/WMMK4LKYHRFVFGEJO6QNTOERVV.json","view_paper":"https://pith.science/paper/WMMK4LKY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.07536&json=true","fetch_graph":"https://pith.science/api/pith-number/WMMK4LKYHRFVFGEJO6QNTOERVV/graph.json","fetch_events":"https://pith.science/api/pith-number/WMMK4LKYHRFVFGEJO6QNTOERVV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WMMK4LKYHRFVFGEJO6QNTOERVV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WMMK4LKYHRFVFGEJO6QNTOERVV/action/storage_attestation","attest_author":"https://pith.science/pith/WMMK4LKYHRFVFGEJO6QNTOERVV/action/author_attestation","sign_citation":"https://pith.science/pith/WMMK4LKYHRFVFGEJO6QNTOERVV/action/citation_signature","submit_replication":"https://pith.science/pith/WMMK4LKYHRFVFGEJO6QNTOERVV/action/replication_record"}},"created_at":"2026-07-05T08:58:41.209657+00:00","updated_at":"2026-07-05T08:58:41.209657+00:00"}