{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:NPEQLCWRS2V63MISOZBGUGQDFX","short_pith_number":"pith:NPEQLCWR","schema_version":"1.0","canonical_sha256":"6bc9058ad196abedb11276426a1a032dc290e385fa08bce9e2fe8f8c4e6022ed","source":{"kind":"arxiv","id":"2402.15745","version":2},"attestation_state":"computed","paper":{"title":"GAOKAO-MM: A Chinese Human-Level Benchmark for Multimodal Models Evaluation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.CL","authors_text":"Xipeng Qiu, Yi Zong","submitted_at":"2024-02-24T06:57:15Z","abstract_excerpt":"The Large Vision-Language Models (LVLMs) have demonstrated great abilities in image perception and language understanding. However, existing multimodal benchmarks focus on primary perception abilities and commonsense knowledge which are insufficient to reflect the comprehensive capabilities of LVLMs. We propose GAOKAO-MM, a multimodal benchmark based on the Chinese College Entrance Examination (GAOKAO), comprising of 8 subjects and 12 types of images, such as diagrams, function graphs, maps and photos. GAOKAO-MM derives from native Chinese context and sets human-level requirements for the mode"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.15745","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-02-24T06:57:15Z","cross_cats_sorted":["cs.AI","cs.CV"],"title_canon_sha256":"505fda9bdfbb3d689de9a1fb58e9a9b0178a0483d9679e42a769624deb3e7640","abstract_canon_sha256":"07138891f3ddb76d07ab95fd33526f898dacbc99afe7ed3dcc6fde8db3ba66ec"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:52:29.495705Z","signature_b64":"zpnyPcGs8m28t2HjidD8zUVs1UPl2snK8Nlpr4lUQDWcppQdc4gCul4KQ8J24fgSGXyAKr+dEcbJfkRq2k3MBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6bc9058ad196abedb11276426a1a032dc290e385fa08bce9e2fe8f8c4e6022ed","last_reissued_at":"2026-07-05T08:52:29.495290Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:52:29.495290Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GAOKAO-MM: A Chinese Human-Level Benchmark for Multimodal Models Evaluation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.CL","authors_text":"Xipeng Qiu, Yi Zong","submitted_at":"2024-02-24T06:57:15Z","abstract_excerpt":"The Large Vision-Language Models (LVLMs) have demonstrated great abilities in image perception and language understanding. However, existing multimodal benchmarks focus on primary perception abilities and commonsense knowledge which are insufficient to reflect the comprehensive capabilities of LVLMs. We propose GAOKAO-MM, a multimodal benchmark based on the Chinese College Entrance Examination (GAOKAO), comprising of 8 subjects and 12 types of images, such as diagrams, function graphs, maps and photos. GAOKAO-MM derives from native Chinese context and sets human-level requirements for the mode"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.15745","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.15745/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.15745","created_at":"2026-07-05T08:52:29.495359+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.15745v2","created_at":"2026-07-05T08:52:29.495359+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.15745","created_at":"2026-07-05T08:52:29.495359+00:00"},{"alias_kind":"pith_short_12","alias_value":"NPEQLCWRS2V6","created_at":"2026-07-05T08:52:29.495359+00:00"},{"alias_kind":"pith_short_16","alias_value":"NPEQLCWRS2V63MIS","created_at":"2026-07-05T08:52:29.495359+00:00"},{"alias_kind":"pith_short_8","alias_value":"NPEQLCWR","created_at":"2026-07-05T08:52:29.495359+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NPEQLCWRS2V63MISOZBGUGQDFX","json":"https://pith.science/pith/NPEQLCWRS2V63MISOZBGUGQDFX.json","graph_json":"https://pith.science/api/pith-number/NPEQLCWRS2V63MISOZBGUGQDFX/graph.json","events_json":"https://pith.science/api/pith-number/NPEQLCWRS2V63MISOZBGUGQDFX/events.json","paper":"https://pith.science/paper/NPEQLCWR"},"agent_actions":{"view_html":"https://pith.science/pith/NPEQLCWRS2V63MISOZBGUGQDFX","download_json":"https://pith.science/pith/NPEQLCWRS2V63MISOZBGUGQDFX.json","view_paper":"https://pith.science/paper/NPEQLCWR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.15745&json=true","fetch_graph":"https://pith.science/api/pith-number/NPEQLCWRS2V63MISOZBGUGQDFX/graph.json","fetch_events":"https://pith.science/api/pith-number/NPEQLCWRS2V63MISOZBGUGQDFX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NPEQLCWRS2V63MISOZBGUGQDFX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NPEQLCWRS2V63MISOZBGUGQDFX/action/storage_attestation","attest_author":"https://pith.science/pith/NPEQLCWRS2V63MISOZBGUGQDFX/action/author_attestation","sign_citation":"https://pith.science/pith/NPEQLCWRS2V63MISOZBGUGQDFX/action/citation_signature","submit_replication":"https://pith.science/pith/NPEQLCWRS2V63MISOZBGUGQDFX/action/replication_record"}},"created_at":"2026-07-05T08:52:29.495359+00:00","updated_at":"2026-07-05T08:52:29.495359+00:00"}