{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:GU6DHDWQ23ECLC7YMDCYU5POC6","short_pith_number":"pith:GU6DHDWQ","schema_version":"1.0","canonical_sha256":"353c338ed0d6c8258bf860c58a75ee17b40d8f443656e508dc19ebdc5c111427","source":{"kind":"arxiv","id":"2409.18142","version":1},"attestation_state":"computed","paper":{"title":"A Survey on Multimodal Benchmarks: In the Era of Large AI Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.MM"],"primary_cat":"cs.AI","authors_text":"Guikun Chen, Hanrong Shi, Jun Xiao, Lin Li, Long Chen","submitted_at":"2024-09-21T15:22:26Z","abstract_excerpt":"The rapid evolution of Multimodal Large Language Models (MLLMs) has brought substantial advancements in artificial intelligence, significantly enhancing the capability to understand and generate multimodal content. While prior studies have largely concentrated on model architectures and training methodologies, a thorough analysis of the benchmarks used for evaluating these models remains underexplored. This survey addresses this gap by systematically reviewing 211 benchmarks that assess MLLMs across four core domains: understanding, reasoning, generation, and application. We provide a detailed"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.18142","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2024-09-21T15:22:26Z","cross_cats_sorted":["cs.MM"],"title_canon_sha256":"9829d12328460dcbd975002c51c0c7d3164e771115dc0a60f59939fd77264a56","abstract_canon_sha256":"99b63901d8e77d6f1d91a52dcfd064369a5f0a1aa27f9510637e5936250bf24c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:12:31.731263Z","signature_b64":"GGeLn/ojdVTH5uSZgtx95R/8Ic2FnLjEASjdHs9qwI9tL2xWQHm1WMBMsxQnyaJhgzM5krpS/RePqJPGsyG3AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"353c338ed0d6c8258bf860c58a75ee17b40d8f443656e508dc19ebdc5c111427","last_reissued_at":"2026-07-05T09:12:31.730810Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:12:31.730810Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Survey on Multimodal Benchmarks: In the Era of Large AI Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.MM"],"primary_cat":"cs.AI","authors_text":"Guikun Chen, Hanrong Shi, Jun Xiao, Lin Li, Long Chen","submitted_at":"2024-09-21T15:22:26Z","abstract_excerpt":"The rapid evolution of Multimodal Large Language Models (MLLMs) has brought substantial advancements in artificial intelligence, significantly enhancing the capability to understand and generate multimodal content. While prior studies have largely concentrated on model architectures and training methodologies, a thorough analysis of the benchmarks used for evaluating these models remains underexplored. This survey addresses this gap by systematically reviewing 211 benchmarks that assess MLLMs across four core domains: understanding, reasoning, generation, and application. We provide a detailed"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.18142","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.18142/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.18142","created_at":"2026-07-05T09:12:31.730870+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.18142v1","created_at":"2026-07-05T09:12:31.730870+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.18142","created_at":"2026-07-05T09:12:31.730870+00:00"},{"alias_kind":"pith_short_12","alias_value":"GU6DHDWQ23EC","created_at":"2026-07-05T09:12:31.730870+00:00"},{"alias_kind":"pith_short_16","alias_value":"GU6DHDWQ23ECLC7Y","created_at":"2026-07-05T09:12:31.730870+00:00"},{"alias_kind":"pith_short_8","alias_value":"GU6DHDWQ","created_at":"2026-07-05T09:12:31.730870+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2502.02871","citing_title":"Position: Multimodal Large Language Models Can Significantly Advance Scientific Reasoning","ref_index":95,"is_internal_anchor":false},{"citing_arxiv_id":"2505.15616","citing_title":"LENS: Multi-level Evaluation of Multimodal Reasoning with Large Language Models","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2411.18279","citing_title":"Large Language Model-Brained GUI Agents: A Survey","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09278","citing_title":"EquiMem: Calibrating Shared Memory in Multi-Agent Debate via Game-Theoretic Equilibrium","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00814","citing_title":"Persistent Visual Memory: Sustaining Perception for Deep Generation in LVLMs","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00814","citing_title":"Persistent Visual Memory: Sustaining Perception for Deep Generation in LVLMs","ref_index":40,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GU6DHDWQ23ECLC7YMDCYU5POC6","json":"https://pith.science/pith/GU6DHDWQ23ECLC7YMDCYU5POC6.json","graph_json":"https://pith.science/api/pith-number/GU6DHDWQ23ECLC7YMDCYU5POC6/graph.json","events_json":"https://pith.science/api/pith-number/GU6DHDWQ23ECLC7YMDCYU5POC6/events.json","paper":"https://pith.science/paper/GU6DHDWQ"},"agent_actions":{"view_html":"https://pith.science/pith/GU6DHDWQ23ECLC7YMDCYU5POC6","download_json":"https://pith.science/pith/GU6DHDWQ23ECLC7YMDCYU5POC6.json","view_paper":"https://pith.science/paper/GU6DHDWQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.18142&json=true","fetch_graph":"https://pith.science/api/pith-number/GU6DHDWQ23ECLC7YMDCYU5POC6/graph.json","fetch_events":"https://pith.science/api/pith-number/GU6DHDWQ23ECLC7YMDCYU5POC6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GU6DHDWQ23ECLC7YMDCYU5POC6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GU6DHDWQ23ECLC7YMDCYU5POC6/action/storage_attestation","attest_author":"https://pith.science/pith/GU6DHDWQ23ECLC7YMDCYU5POC6/action/author_attestation","sign_citation":"https://pith.science/pith/GU6DHDWQ23ECLC7YMDCYU5POC6/action/citation_signature","submit_replication":"https://pith.science/pith/GU6DHDWQ23ECLC7YMDCYU5POC6/action/replication_record"}},"created_at":"2026-07-05T09:12:31.730870+00:00","updated_at":"2026-07-05T09:12:31.730870+00:00"}