{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:G5AHLLLK5QL34AC4H3ZN336RMQ","short_pith_number":"pith:G5AHLLLK","schema_version":"1.0","canonical_sha256":"374075ad6aec17be005c3ef2ddefd16417fd51b12d4f87bffc1059d3abb90477","source":{"kind":"arxiv","id":"2405.15638","version":2},"attestation_state":"computed","paper":{"title":"M4U: Evaluating Multilingual Understanding and Reasoning for Large Multimodal Models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Bin Zhang, Chuyan Xiong, Hongyu Wang, Jialin Li, Jiayu Xu, Ruiping Wang, Senwei Xie, Xilin Chen, Zhaojie Xie","submitted_at":"2024-05-24T15:25:28Z","abstract_excerpt":"Multilingual capability is an essential aspect for large multimodal models, since they are usually deployed across various countries and languages. However, most existing benchmarks for multilingual multimodal reasoning struggle to differentiate between models of varying performance; even language models without visual capabilities can easily achieve high scores. This leaves a comprehensive evaluation of leading multilingual multimodal models largely unexplored. In this work, we introduce M4U, a novel and challenging benchmark for assessing the capability of multi-discipline multilingual multi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.15638","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-05-24T15:25:28Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"ad8e45563e9b8c91a9677bdb9744f606e3587c7aeb2b73c0a9a96116fed19606","abstract_canon_sha256":"ea21b0c67684b1faf4c5bfe94f9a293d8de19669f00d2234d7422d044ceb279d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:53:45.170270Z","signature_b64":"qeXK25hTQVIjxUyy6YaZQtCrWzn6EssT0NYVDI8r121XhNIbxizA594gSL689Lsg9IDmWWyX59V1YBSgyU2FBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"374075ad6aec17be005c3ef2ddefd16417fd51b12d4f87bffc1059d3abb90477","last_reissued_at":"2026-07-05T10:53:45.169759Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:53:45.169759Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"M4U: Evaluating Multilingual Understanding and Reasoning for Large Multimodal Models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Bin Zhang, Chuyan Xiong, Hongyu Wang, Jialin Li, Jiayu Xu, Ruiping Wang, Senwei Xie, Xilin Chen, Zhaojie Xie","submitted_at":"2024-05-24T15:25:28Z","abstract_excerpt":"Multilingual capability is an essential aspect for large multimodal models, since they are usually deployed across various countries and languages. However, most existing benchmarks for multilingual multimodal reasoning struggle to differentiate between models of varying performance; even language models without visual capabilities can easily achieve high scores. This leaves a comprehensive evaluation of leading multilingual multimodal models largely unexplored. In this work, we introduce M4U, a novel and challenging benchmark for assessing the capability of multi-discipline multilingual multi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.15638","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.15638/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.15638","created_at":"2026-07-05T10:53:45.169821+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.15638v2","created_at":"2026-07-05T10:53:45.169821+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.15638","created_at":"2026-07-05T10:53:45.169821+00:00"},{"alias_kind":"pith_short_12","alias_value":"G5AHLLLK5QL3","created_at":"2026-07-05T10:53:45.169821+00:00"},{"alias_kind":"pith_short_16","alias_value":"G5AHLLLK5QL34AC4","created_at":"2026-07-05T10:53:45.169821+00:00"},{"alias_kind":"pith_short_8","alias_value":"G5AHLLLK","created_at":"2026-07-05T10:53:45.169821+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11470","citing_title":"The Periodic Table of LLM Reasoning: A Structured Survey of Reasoning Paradigms, Methods, and Failure Modes","ref_index":239,"is_internal_anchor":false},{"citing_arxiv_id":"2502.02871","citing_title":"Position: Multimodal Large Language Models Can Significantly Advance Scientific Reasoning","ref_index":190,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/G5AHLLLK5QL34AC4H3ZN336RMQ","json":"https://pith.science/pith/G5AHLLLK5QL34AC4H3ZN336RMQ.json","graph_json":"https://pith.science/api/pith-number/G5AHLLLK5QL34AC4H3ZN336RMQ/graph.json","events_json":"https://pith.science/api/pith-number/G5AHLLLK5QL34AC4H3ZN336RMQ/events.json","paper":"https://pith.science/paper/G5AHLLLK"},"agent_actions":{"view_html":"https://pith.science/pith/G5AHLLLK5QL34AC4H3ZN336RMQ","download_json":"https://pith.science/pith/G5AHLLLK5QL34AC4H3ZN336RMQ.json","view_paper":"https://pith.science/paper/G5AHLLLK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.15638&json=true","fetch_graph":"https://pith.science/api/pith-number/G5AHLLLK5QL34AC4H3ZN336RMQ/graph.json","fetch_events":"https://pith.science/api/pith-number/G5AHLLLK5QL34AC4H3ZN336RMQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/G5AHLLLK5QL34AC4H3ZN336RMQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/G5AHLLLK5QL34AC4H3ZN336RMQ/action/storage_attestation","attest_author":"https://pith.science/pith/G5AHLLLK5QL34AC4H3ZN336RMQ/action/author_attestation","sign_citation":"https://pith.science/pith/G5AHLLLK5QL34AC4H3ZN336RMQ/action/citation_signature","submit_replication":"https://pith.science/pith/G5AHLLLK5QL34AC4H3ZN336RMQ/action/replication_record"}},"created_at":"2026-07-05T10:53:45.169821+00:00","updated_at":"2026-07-05T10:53:45.169821+00:00"}