{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:HLAVBY5AMPMCPJYHBRJUUESH7S","short_pith_number":"pith:HLAVBY5A","schema_version":"1.0","canonical_sha256":"3ac150e3a063d827a7070c534a1247fc8166f8a4213cf3f5772a916f866a3d34","source":{"kind":"arxiv","id":"2408.08632","version":2},"attestation_state":"computed","paper":{"title":"A Survey on Benchmarks of Multimodal Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.CL","authors_text":"Chaoyou Fu, Chengjie Wang, Ding Qi, Hao Fei, Jian Li, Meng Luo, Ming Dai, Min Xia, Wankou Yang, Weiheng Lu, Yabiao Wang, Ying Tai, Yizhang Jin, Zhenye Gan","submitted_at":"2024-08-16T09:52:02Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) are gaining increasing popularity in both academia and industry due to their remarkable performance in various applications such as visual question answering, visual perception, understanding, and reasoning. Over the past few years, significant efforts have been made to examine MLLMs from multiple perspectives. This paper presents a comprehensive review of 200 benchmarks and evaluations for MLLMs, focusing on (1)perception and understanding, (2)cognition and reasoning, (3)specific domains, (4)key capabilities, and (5)other modalities. Finally, we discus"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.08632","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-08-16T09:52:02Z","cross_cats_sorted":["cs.AI","cs.CV"],"title_canon_sha256":"817d50fd02e0b0cef276840729a5812456025ee590012e8219ce5a04f012122c","abstract_canon_sha256":"abc53a326e61881018b120dec37ce7a7959e819e07d16bd751e59fe05cca9187"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:03:55.319899Z","signature_b64":"CtCsu2hWO1Qfr/OxP07NRxpnAx5bzHy104UTXLasSMFyoRysNQ+LqslSxPh7r0IsNLvd03oNAix9ZvoJsKppBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3ac150e3a063d827a7070c534a1247fc8166f8a4213cf3f5772a916f866a3d34","last_reissued_at":"2026-07-05T09:03:55.319476Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:03:55.319476Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Survey on Benchmarks of Multimodal Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.CL","authors_text":"Chaoyou Fu, Chengjie Wang, Ding Qi, Hao Fei, Jian Li, Meng Luo, Ming Dai, Min Xia, Wankou Yang, Weiheng Lu, Yabiao Wang, Ying Tai, Yizhang Jin, Zhenye Gan","submitted_at":"2024-08-16T09:52:02Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) are gaining increasing popularity in both academia and industry due to their remarkable performance in various applications such as visual question answering, visual perception, understanding, and reasoning. Over the past few years, significant efforts have been made to examine MLLMs from multiple perspectives. This paper presents a comprehensive review of 200 benchmarks and evaluations for MLLMs, focusing on (1)perception and understanding, (2)cognition and reasoning, (3)specific domains, (4)key capabilities, and (5)other modalities. Finally, we discus"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.08632","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.08632/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.08632","created_at":"2026-07-05T09:03:55.319531+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.08632v2","created_at":"2026-07-05T09:03:55.319531+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.08632","created_at":"2026-07-05T09:03:55.319531+00:00"},{"alias_kind":"pith_short_12","alias_value":"HLAVBY5AMPMC","created_at":"2026-07-05T09:03:55.319531+00:00"},{"alias_kind":"pith_short_16","alias_value":"HLAVBY5AMPMCPJYH","created_at":"2026-07-05T09:03:55.319531+00:00"},{"alias_kind":"pith_short_8","alias_value":"HLAVBY5A","created_at":"2026-07-05T09:03:55.319531+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22437","citing_title":"MMGist: A Comprehensive Multimodal Benchmark for 2027","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06217","citing_title":"DisasterBench: A Multimodal Benchmark for UAV-Based Disaster Response in Complex Environments","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04349","citing_title":"MorphoQuant: Modality-Aware Quantization for Omni-modal Large Language Models","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2502.02871","citing_title":"Position: Multimodal Large Language Models Can Significantly Advance Scientific Reasoning","ref_index":93,"is_internal_anchor":false},{"citing_arxiv_id":"2503.23733","citing_title":"AdaMMS: Model Merging for Heterogeneous Multimodal Large Language Models with Unsupervised Coefficient Optimization","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2603.11689","citing_title":"Explicit Logic Channel for Validation and Enhancement of MLLMs on Zero-Shot Tasks","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15777","citing_title":"SaaS-Bench: Can Computer-Use Agents Leverage Real-World SaaS to Solve Professional Workflows?","ref_index":87,"is_internal_anchor":false},{"citing_arxiv_id":"2508.08508","citing_title":"Re:Verse -- Can Your VLM Read a Manga?","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03903","citing_title":"CC-OCR V2: Benchmarking Large Multimodal Models for Literacy in Real-world Document Processing","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19083","citing_title":"ProjLens: Unveiling the Role of Projectors in Multimodal Model Safety","ref_index":144,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08615","citing_title":"MARINER: A 3E-Driven Benchmark for Fine-Grained Perception and Complex Reasoning in Open-Water Environments","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14629","citing_title":"Switch-KD: Visual-Switch Knowledge Distillation for Vision-Language Models","ref_index":25,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HLAVBY5AMPMCPJYHBRJUUESH7S","json":"https://pith.science/pith/HLAVBY5AMPMCPJYHBRJUUESH7S.json","graph_json":"https://pith.science/api/pith-number/HLAVBY5AMPMCPJYHBRJUUESH7S/graph.json","events_json":"https://pith.science/api/pith-number/HLAVBY5AMPMCPJYHBRJUUESH7S/events.json","paper":"https://pith.science/paper/HLAVBY5A"},"agent_actions":{"view_html":"https://pith.science/pith/HLAVBY5AMPMCPJYHBRJUUESH7S","download_json":"https://pith.science/pith/HLAVBY5AMPMCPJYHBRJUUESH7S.json","view_paper":"https://pith.science/paper/HLAVBY5A","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.08632&json=true","fetch_graph":"https://pith.science/api/pith-number/HLAVBY5AMPMCPJYHBRJUUESH7S/graph.json","fetch_events":"https://pith.science/api/pith-number/HLAVBY5AMPMCPJYHBRJUUESH7S/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HLAVBY5AMPMCPJYHBRJUUESH7S/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HLAVBY5AMPMCPJYHBRJUUESH7S/action/storage_attestation","attest_author":"https://pith.science/pith/HLAVBY5AMPMCPJYHBRJUUESH7S/action/author_attestation","sign_citation":"https://pith.science/pith/HLAVBY5AMPMCPJYHBRJUUESH7S/action/citation_signature","submit_replication":"https://pith.science/pith/HLAVBY5AMPMCPJYHBRJUUESH7S/action/replication_record"}},"created_at":"2026-07-05T09:03:55.319531+00:00","updated_at":"2026-07-05T09:03:55.319531+00:00"}