{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:5GUTQ2N6CXGF4SNJZQT37CGG6F","short_pith_number":"pith:5GUTQ2N6","schema_version":"1.0","canonical_sha256":"e9a93869be15cc5e49a9cc27bf88c6f1723d3340c182d65a6d4a023c96d31f5e","source":{"kind":"arxiv","id":"2408.15769","version":1},"attestation_state":"computed","paper":{"title":"A Survey on Evaluation of Multimodal Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Jiaxing Huang, Jingyi Zhang","submitted_at":"2024-08-28T13:05:55Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) mimic human perception and reasoning system by integrating powerful Large Language Models (LLMs) with various modality encoders (e.g., vision, audio), positioning LLMs as the \"brain\" and various modality encoders as sensory organs. This framework endows MLLMs with human-like capabilities, and suggests a potential pathway towards achieving artificial general intelligence (AGI). With the emergence of all-round MLLMs like GPT-4V and Gemini, a multitude of evaluation methods have been developed to assess their capabilities across different dimensions. This "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.15769","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-08-28T13:05:55Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"e5d09d22a3ca867b8c94e33ed3021f5a721f9f583677c727d4b10cd61beb89c0","abstract_canon_sha256":"fb367b9b15f9a6169bf8e9549aa1742ebec4b3d044ca942bfec0e5d82b824a50"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:00:17.168949Z","signature_b64":"HePUzNx9d40hNAgM/x/ugynpUTWIy8X9YirZCh8evVQLJ2LwNOje/YEz5LbdIgp2FtvFxbZmtPsPHiFMzbTfBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e9a93869be15cc5e49a9cc27bf88c6f1723d3340c182d65a6d4a023c96d31f5e","last_reissued_at":"2026-07-05T09:00:17.168470Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:00:17.168470Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Survey on Evaluation of Multimodal Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Jiaxing Huang, Jingyi Zhang","submitted_at":"2024-08-28T13:05:55Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) mimic human perception and reasoning system by integrating powerful Large Language Models (LLMs) with various modality encoders (e.g., vision, audio), positioning LLMs as the \"brain\" and various modality encoders as sensory organs. This framework endows MLLMs with human-like capabilities, and suggests a potential pathway towards achieving artificial general intelligence (AGI). With the emergence of all-round MLLMs like GPT-4V and Gemini, a multitude of evaluation methods have been developed to assess their capabilities across different dimensions. This "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.15769","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.15769/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.15769","created_at":"2026-07-05T09:00:17.168525+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.15769v1","created_at":"2026-07-05T09:00:17.168525+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.15769","created_at":"2026-07-05T09:00:17.168525+00:00"},{"alias_kind":"pith_short_12","alias_value":"5GUTQ2N6CXGF","created_at":"2026-07-05T09:00:17.168525+00:00"},{"alias_kind":"pith_short_16","alias_value":"5GUTQ2N6CXGF4SNJ","created_at":"2026-07-05T09:00:17.168525+00:00"},{"alias_kind":"pith_short_8","alias_value":"5GUTQ2N6","created_at":"2026-07-05T09:00:17.168525+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2603.16284","citing_title":"Locate-then-Sparsify: Attribution Guided Sparse Strategy for Visual Hallucination Mitigation","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2506.09082","citing_title":"AVA-Bench: Atomic Visual Ability Benchmark for Vision Foundation Models","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2411.18279","citing_title":"Large Language Model-Brained GUI Agents: A Survey","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2503.09567","citing_title":"Towards Reasoning Era: A Survey of Long Chain-of-Thought for Reasoning Large Language Models","ref_index":289,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06115","citing_title":"CrossCult-KIBench: A Benchmark for Cross-Cultural Knowledge Insertion in MLLMs","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08703","citing_title":"QoS-QoE Translation with Large Language Model","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06115","citing_title":"CrossCult-KIBench: A Benchmark for Cross-Cultural Knowledge Insertion in MLLMs","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5GUTQ2N6CXGF4SNJZQT37CGG6F","json":"https://pith.science/pith/5GUTQ2N6CXGF4SNJZQT37CGG6F.json","graph_json":"https://pith.science/api/pith-number/5GUTQ2N6CXGF4SNJZQT37CGG6F/graph.json","events_json":"https://pith.science/api/pith-number/5GUTQ2N6CXGF4SNJZQT37CGG6F/events.json","paper":"https://pith.science/paper/5GUTQ2N6"},"agent_actions":{"view_html":"https://pith.science/pith/5GUTQ2N6CXGF4SNJZQT37CGG6F","download_json":"https://pith.science/pith/5GUTQ2N6CXGF4SNJZQT37CGG6F.json","view_paper":"https://pith.science/paper/5GUTQ2N6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.15769&json=true","fetch_graph":"https://pith.science/api/pith-number/5GUTQ2N6CXGF4SNJZQT37CGG6F/graph.json","fetch_events":"https://pith.science/api/pith-number/5GUTQ2N6CXGF4SNJZQT37CGG6F/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5GUTQ2N6CXGF4SNJZQT37CGG6F/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5GUTQ2N6CXGF4SNJZQT37CGG6F/action/storage_attestation","attest_author":"https://pith.science/pith/5GUTQ2N6CXGF4SNJZQT37CGG6F/action/author_attestation","sign_citation":"https://pith.science/pith/5GUTQ2N6CXGF4SNJZQT37CGG6F/action/citation_signature","submit_replication":"https://pith.science/pith/5GUTQ2N6CXGF4SNJZQT37CGG6F/action/replication_record"}},"created_at":"2026-07-05T09:00:17.168525+00:00","updated_at":"2026-07-05T09:00:17.168525+00:00"}