{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:5R4KTUKLYKEHGJFN5GPPN63JSO","short_pith_number":"pith:5R4KTUKL","schema_version":"1.0","canonical_sha256":"ec78a9d14bc2887324ade99ef6fb6993a17cfd2d81718efe33f1fbdd84b3c169","source":{"kind":"arxiv","id":"2505.16459","version":4},"attestation_state":"computed","paper":{"title":"MMLU-Reason: Benchmarking Multi-Task Multi-modal Language Understanding and Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Chaoran Hu, Guiyao Tie, Lichao Sun, Mengqu Sun, Pan Zhou, Ruihang Zhang, Sizhe Zhang, Tianhe Gu, Xueyang Zhou, Yan Zhang","submitted_at":"2025-05-22T09:41:55Z","abstract_excerpt":"Recent advances in Multi-Modal Large Language Models (MLLMs) have enabled unified processing of language, vision, and structured inputs, opening the door to complex tasks such as logical deduction, spatial reasoning, and scientific analysis. Despite their promise, the reasoning capabilities of MLLMs, particularly those augmented with intermediate thinking traces (MLLMs-T), remain poorly understood and lack standardized evaluation benchmarks. Existing work focuses primarily on perception or final answer correctness, offering limited insight into how models reason or fail across modalities. To a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.16459","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-05-22T09:41:55Z","cross_cats_sorted":[],"title_canon_sha256":"53b454dc45b98d077d76619c5dd315def9d2557333a84ae8b51a06e0d80ce568","abstract_canon_sha256":"1365ac7284cb49bc8cdc9e08c3462c388e7ebb73ac3253917af6672bb3b929bc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:30:23.236246Z","signature_b64":"zZFt4kNwtmKvvCkZ/eGjJZR4iM2I+Wm/ORFDF3tqB716Czxi2ZZF+fp5YvdDDRka74xiHhZOlJtYjO4NRwetDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ec78a9d14bc2887324ade99ef6fb6993a17cfd2d81718efe33f1fbdd84b3c169","last_reissued_at":"2026-07-05T11:30:23.235849Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:30:23.235849Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MMLU-Reason: Benchmarking Multi-Task Multi-modal Language Understanding and Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Chaoran Hu, Guiyao Tie, Lichao Sun, Mengqu Sun, Pan Zhou, Ruihang Zhang, Sizhe Zhang, Tianhe Gu, Xueyang Zhou, Yan Zhang","submitted_at":"2025-05-22T09:41:55Z","abstract_excerpt":"Recent advances in Multi-Modal Large Language Models (MLLMs) have enabled unified processing of language, vision, and structured inputs, opening the door to complex tasks such as logical deduction, spatial reasoning, and scientific analysis. Despite their promise, the reasoning capabilities of MLLMs, particularly those augmented with intermediate thinking traces (MLLMs-T), remain poorly understood and lack standardized evaluation benchmarks. Existing work focuses primarily on perception or final answer correctness, offering limited insight into how models reason or fail across modalities. To a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.16459","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.16459/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.16459","created_at":"2026-07-05T11:30:23.235899+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.16459v4","created_at":"2026-07-05T11:30:23.235899+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.16459","created_at":"2026-07-05T11:30:23.235899+00:00"},{"alias_kind":"pith_short_12","alias_value":"5R4KTUKLYKEH","created_at":"2026-07-05T11:30:23.235899+00:00"},{"alias_kind":"pith_short_16","alias_value":"5R4KTUKLYKEHGJFN","created_at":"2026-07-05T11:30:23.235899+00:00"},{"alias_kind":"pith_short_8","alias_value":"5R4KTUKL","created_at":"2026-07-05T11:30:23.235899+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2608.05631","citing_title":"ChronoVision: Temporal Reasoning via Latent State Reconstruction","ref_index":2009,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5R4KTUKLYKEHGJFN5GPPN63JSO","json":"https://pith.science/pith/5R4KTUKLYKEHGJFN5GPPN63JSO.json","graph_json":"https://pith.science/api/pith-number/5R4KTUKLYKEHGJFN5GPPN63JSO/graph.json","events_json":"https://pith.science/api/pith-number/5R4KTUKLYKEHGJFN5GPPN63JSO/events.json","paper":"https://pith.science/paper/5R4KTUKL"},"agent_actions":{"view_html":"https://pith.science/pith/5R4KTUKLYKEHGJFN5GPPN63JSO","download_json":"https://pith.science/pith/5R4KTUKLYKEHGJFN5GPPN63JSO.json","view_paper":"https://pith.science/paper/5R4KTUKL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.16459&json=true","fetch_graph":"https://pith.science/api/pith-number/5R4KTUKLYKEHGJFN5GPPN63JSO/graph.json","fetch_events":"https://pith.science/api/pith-number/5R4KTUKLYKEHGJFN5GPPN63JSO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5R4KTUKLYKEHGJFN5GPPN63JSO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5R4KTUKLYKEHGJFN5GPPN63JSO/action/storage_attestation","attest_author":"https://pith.science/pith/5R4KTUKLYKEHGJFN5GPPN63JSO/action/author_attestation","sign_citation":"https://pith.science/pith/5R4KTUKLYKEHGJFN5GPPN63JSO/action/citation_signature","submit_replication":"https://pith.science/pith/5R4KTUKLYKEHGJFN5GPPN63JSO/action/replication_record"}},"created_at":"2026-07-05T11:30:23.235899+00:00","updated_at":"2026-07-05T11:30:23.235899+00:00"}