{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:K6XCMMNJ272IUWLT22SXL56XNJ","short_pith_number":"pith:K6XCMMNJ","schema_version":"1.0","canonical_sha256":"57ae2631a9d7f48a5973d6a575f7d76a4102ac80bd74ffc2dbe748d77913496d","source":{"kind":"arxiv","id":"2411.15296","version":2},"attestation_state":"computed","paper":{"title":"MME-Survey: A Comprehensive Survey on Evaluation of Multimodal LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Bo Li, Caifeng Shan, Chaoyou Fu, Haodong Duan, Liang Wang, Ran He, Shukang Yin, Sirui Zhao, Xing Sun, Xinyu Fang, Yi-Fan Zhang, Ziwei Liu","submitted_at":"2024-11-22T18:59:54Z","abstract_excerpt":"As a prominent direction of Artificial General Intelligence (AGI), Multimodal Large Language Models (MLLMs) have garnered increased attention from both industry and academia. Building upon pre-trained LLMs, this family of models further develops multimodal perception and reasoning capabilities that are impressive, such as writing code given a flow chart or creating stories based on an image. In the development process, evaluation is critical since it provides intuitive feedback and guidance on improving models. Distinct from the traditional train-eval-test paradigm that only favors a single ta"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.15296","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-11-22T18:59:54Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"b25bc715d5aec5d915046410ac4f820fc334aa1036b95f02e4bb1e96e31ba311","abstract_canon_sha256":"5efb875bca14cb2ed0ed3fc70f9559195658fe765790637024bd82e9778f379c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:45:58.528518Z","signature_b64":"26pNU7Wy/RVLFNM/yR1OqT7X+vQWx45Iyr/9ADNpda5MFePG8f+Hn+otzYi8/BWSh4Rh1Y/x+RJXm52wF00XAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"57ae2631a9d7f48a5973d6a575f7d76a4102ac80bd74ffc2dbe748d77913496d","last_reissued_at":"2026-07-05T09:45:58.527908Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:45:58.527908Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MME-Survey: A Comprehensive Survey on Evaluation of Multimodal LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Bo Li, Caifeng Shan, Chaoyou Fu, Haodong Duan, Liang Wang, Ran He, Shukang Yin, Sirui Zhao, Xing Sun, Xinyu Fang, Yi-Fan Zhang, Ziwei Liu","submitted_at":"2024-11-22T18:59:54Z","abstract_excerpt":"As a prominent direction of Artificial General Intelligence (AGI), Multimodal Large Language Models (MLLMs) have garnered increased attention from both industry and academia. Building upon pre-trained LLMs, this family of models further develops multimodal perception and reasoning capabilities that are impressive, such as writing code given a flow chart or creating stories based on an image. In the development process, evaluation is critical since it provides intuitive feedback and guidance on improving models. Distinct from the traditional train-eval-test paradigm that only favors a single ta"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.15296","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.15296/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.15296","created_at":"2026-07-05T09:45:58.527997+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.15296v2","created_at":"2026-07-05T09:45:58.527997+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.15296","created_at":"2026-07-05T09:45:58.527997+00:00"},{"alias_kind":"pith_short_12","alias_value":"K6XCMMNJ272I","created_at":"2026-07-05T09:45:58.527997+00:00"},{"alias_kind":"pith_short_16","alias_value":"K6XCMMNJ272IUWLT","created_at":"2026-07-05T09:45:58.527997+00:00"},{"alias_kind":"pith_short_8","alias_value":"K6XCMMNJ","created_at":"2026-07-05T09:45:58.527997+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":15,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.27446","citing_title":"Causal Connections: Leveraging Multilingual Fine-Tuning for Financial QA@FinCausal 2026","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27316","citing_title":"LLM-Based Examination of Eligibility Criteria from Securities Prospectuses at the German Central Bank","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2502.02871","citing_title":"Position: Multimodal Large Language Models Can Significantly Advance Scientific Reasoning","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2505.15616","citing_title":"LENS: Multi-level Evaluation of Multimodal Reasoning with Large Language Models","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2602.13294","citing_title":"VisPhyWorld: Probing Physical Reasoning via Code-Driven Video Reconstruction","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2603.11689","citing_title":"Explicit Logic Channel for Validation and Enhancement of MLLMs on Zero-Shot Tasks","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21479","citing_title":"WikiVQABench: A Knowledge-Grounded Visual Question Answering Benchmark from Wikipedia and Wikidata","ref_index":115,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18018","citing_title":"See What I Mean: Aligning Vision and Language Representations for Video Fine-grained Object Understanding","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18115","citing_title":"WinTok: A Win-Win Hybrid Tokenizer via Decomposing Visual Understanding and Generation with Transferable Tokens","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2510.21828","citing_title":"Structured and Abstractive Reasoning on Multi-modal Relational Knowledge Images","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2603.11665","citing_title":"Multi-Task Reinforcement Learning for Enhanced Multimodal LLM-as-a-Judge","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11716","citing_title":"SafeSteer: A Decoding-level Defense Mechanism for Multimodal Large Language Models","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03903","citing_title":"CC-OCR V2: Benchmarking Large Multimodal Models for Literacy in Real-world Document Processing","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01359","citing_title":"Structural Ranking of the Cognitive Plausibility of Computational Models of Analogy and Metaphors with the Minimal Cognitive Grid","ref_index":196,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12424","citing_title":"Decoding by Perturbation: Mitigating MLLM Hallucinations via Dynamic Textual Perturbation","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/K6XCMMNJ272IUWLT22SXL56XNJ","json":"https://pith.science/pith/K6XCMMNJ272IUWLT22SXL56XNJ.json","graph_json":"https://pith.science/api/pith-number/K6XCMMNJ272IUWLT22SXL56XNJ/graph.json","events_json":"https://pith.science/api/pith-number/K6XCMMNJ272IUWLT22SXL56XNJ/events.json","paper":"https://pith.science/paper/K6XCMMNJ"},"agent_actions":{"view_html":"https://pith.science/pith/K6XCMMNJ272IUWLT22SXL56XNJ","download_json":"https://pith.science/pith/K6XCMMNJ272IUWLT22SXL56XNJ.json","view_paper":"https://pith.science/paper/K6XCMMNJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.15296&json=true","fetch_graph":"https://pith.science/api/pith-number/K6XCMMNJ272IUWLT22SXL56XNJ/graph.json","fetch_events":"https://pith.science/api/pith-number/K6XCMMNJ272IUWLT22SXL56XNJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/K6XCMMNJ272IUWLT22SXL56XNJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/K6XCMMNJ272IUWLT22SXL56XNJ/action/storage_attestation","attest_author":"https://pith.science/pith/K6XCMMNJ272IUWLT22SXL56XNJ/action/author_attestation","sign_citation":"https://pith.science/pith/K6XCMMNJ272IUWLT22SXL56XNJ/action/citation_signature","submit_replication":"https://pith.science/pith/K6XCMMNJ272IUWLT22SXL56XNJ/action/replication_record"}},"created_at":"2026-07-05T09:45:58.527997+00:00","updated_at":"2026-07-05T09:45:58.527997+00:00"}