{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:5D6NBOWVVAELFYOGC4ZG3BXS6L","short_pith_number":"pith:5D6NBOWV","schema_version":"1.0","canonical_sha256":"e8fcd0bad5a808b2e1c617326d86f2f2c992b0c8f8df4b7ac6159d15b14b4091","source":{"kind":"arxiv","id":"2402.04788","version":3},"attestation_state":"computed","paper":{"title":"MLLM-as-a-Judge: Assessing Multimodal LLM-as-a-Judge with Vision-Language Benchmark","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.CL","authors_text":"Dongping Chen, Huichi Zhou, Lichao Sun, Pan Zhou, Qihui Zhang, Ruoxi Chen, Shilin Zhang, Yaochen Wang, Yao Wan, Yinuo Liu","submitted_at":"2024-02-07T12:28:32Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) have gained significant attention recently, showing remarkable potential in artificial general intelligence. However, assessing the utility of MLLMs presents considerable challenges, primarily due to the absence of multimodal benchmarks that align with human preferences. Drawing inspiration from the concept of LLM-as-a-Judge within LLMs, this paper introduces a novel benchmark, termed MLLM-as-a-Judge, to assess the ability of MLLMs in assisting judges across diverse modalities, encompassing three distinct tasks: Scoring Evaluation, Pair Comparison, and "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.04788","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-02-07T12:28:32Z","cross_cats_sorted":["cs.AI","cs.CV"],"title_canon_sha256":"4c27951db34782545a9605ba27efe71dd1ae4fa9d2989c2aebbff4953bad4f67","abstract_canon_sha256":"9c3b9bbbe875099266571b5b4f9cd2236c62b8d7f16dbc37bd367d523f670988"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:30:08.211023Z","signature_b64":"HbcJxE6O3S2fVfIhheoR4kL41LkaQrMifVmt0yjBSevXcUrQMh+5adingvmkQSFd42KpkDtUiB6ZLvXKSzfRAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e8fcd0bad5a808b2e1c617326d86f2f2c992b0c8f8df4b7ac6159d15b14b4091","last_reissued_at":"2026-07-05T08:30:08.210537Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:30:08.210537Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MLLM-as-a-Judge: Assessing Multimodal LLM-as-a-Judge with Vision-Language Benchmark","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.CL","authors_text":"Dongping Chen, Huichi Zhou, Lichao Sun, Pan Zhou, Qihui Zhang, Ruoxi Chen, Shilin Zhang, Yaochen Wang, Yao Wan, Yinuo Liu","submitted_at":"2024-02-07T12:28:32Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) have gained significant attention recently, showing remarkable potential in artificial general intelligence. However, assessing the utility of MLLMs presents considerable challenges, primarily due to the absence of multimodal benchmarks that align with human preferences. Drawing inspiration from the concept of LLM-as-a-Judge within LLMs, this paper introduces a novel benchmark, termed MLLM-as-a-Judge, to assess the ability of MLLMs in assisting judges across diverse modalities, encompassing three distinct tasks: Scoring Evaluation, Pair Comparison, and "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.04788","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.04788/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.04788","created_at":"2026-07-05T08:30:08.210591+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.04788v3","created_at":"2026-07-05T08:30:08.210591+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.04788","created_at":"2026-07-05T08:30:08.210591+00:00"},{"alias_kind":"pith_short_12","alias_value":"5D6NBOWVVAEL","created_at":"2026-07-05T08:30:08.210591+00:00"},{"alias_kind":"pith_short_16","alias_value":"5D6NBOWVVAELFYOG","created_at":"2026-07-05T08:30:08.210591+00:00"},{"alias_kind":"pith_short_8","alias_value":"5D6NBOWV","created_at":"2026-07-05T08:30:08.210591+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05391","citing_title":"LLM-as-a-Verifier: A General-Purpose Verification Framework","ref_index":70,"is_internal_anchor":true},{"citing_arxiv_id":"2606.25432","citing_title":"Brevity is the Soul of Inference Efficiency: Inducing Concision in VLMs via Data Curation","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04773","citing_title":"NextMotionQA: Benchmarking and Judging Human Motion Understanding with Vision-Language Models","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2606.25432","citing_title":"Brevity is the Soul of Inference Efficiency: Inducing Concision in VLMs via Data Curation","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15300","citing_title":"Deep Pre-Alignment for VLMs","ref_index":75,"is_internal_anchor":false},{"citing_arxiv_id":"2412.05579","citing_title":"LLMs-as-Judges: A Comprehensive Survey on LLM-based Evaluation Methods","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2502.18864","citing_title":"Towards an AI co-scientist","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05955","citing_title":"Does Pass Rate Tell the Whole Story? Evaluating Design Constraint Compliance in LLM-based Issue Resolution","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5D6NBOWVVAELFYOGC4ZG3BXS6L","json":"https://pith.science/pith/5D6NBOWVVAELFYOGC4ZG3BXS6L.json","graph_json":"https://pith.science/api/pith-number/5D6NBOWVVAELFYOGC4ZG3BXS6L/graph.json","events_json":"https://pith.science/api/pith-number/5D6NBOWVVAELFYOGC4ZG3BXS6L/events.json","paper":"https://pith.science/paper/5D6NBOWV"},"agent_actions":{"view_html":"https://pith.science/pith/5D6NBOWVVAELFYOGC4ZG3BXS6L","download_json":"https://pith.science/pith/5D6NBOWVVAELFYOGC4ZG3BXS6L.json","view_paper":"https://pith.science/paper/5D6NBOWV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.04788&json=true","fetch_graph":"https://pith.science/api/pith-number/5D6NBOWVVAELFYOGC4ZG3BXS6L/graph.json","fetch_events":"https://pith.science/api/pith-number/5D6NBOWVVAELFYOGC4ZG3BXS6L/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5D6NBOWVVAELFYOGC4ZG3BXS6L/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5D6NBOWVVAELFYOGC4ZG3BXS6L/action/storage_attestation","attest_author":"https://pith.science/pith/5D6NBOWVVAELFYOGC4ZG3BXS6L/action/author_attestation","sign_citation":"https://pith.science/pith/5D6NBOWVVAELFYOGC4ZG3BXS6L/action/citation_signature","submit_replication":"https://pith.science/pith/5D6NBOWVVAELFYOGC4ZG3BXS6L/action/replication_record"}},"created_at":"2026-07-05T08:30:08.210591+00:00","updated_at":"2026-07-05T08:30:08.210591+00:00"}