{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:JD52WJBDHEAOFMSFFNCWLEWRCF","short_pith_number":"pith:JD52WJBD","schema_version":"1.0","canonical_sha256":"48fbab24233900e2b2452b456592d1116b33945c07b59d0b700a214af0a92230","source":{"kind":"arxiv","id":"2503.05977","version":1},"attestation_state":"computed","paper":{"title":"Is Your Video Language Model a Reliable Judge?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Ming Liu, Wensheng Zhang","submitted_at":"2025-03-07T23:17:59Z","abstract_excerpt":"As video language models (VLMs) gain more applications in various scenarios, the need for robust and scalable evaluation of their performance becomes increasingly critical. The traditional human expert-based evaluation of VLMs has limitations in consistency and scalability, which sparked interest in automatic methods such as employing VLMs to evaluate VLMs. However, the reliability of VLMs as judges remains underexplored. Existing methods often rely on a single VLM as the evaluator. However, this approach can be unreliable or biased because such a model may lack the ability to fully understand"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.05977","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-03-07T23:17:59Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"80489093f3143662e76060de067eee3b26508092c929e593039240e1afe5ccb1","abstract_canon_sha256":"8538a7f85a8457fdcff42d5314b4b0ea2a8322aad9f9502b58c6c2ee8872adbb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:26:50.470173Z","signature_b64":"0QuxnUQ5tYfksq/Kco0MzZd74iTJPJRsJn7XSTE5p3dz7k8/WkBCNYhXTCSMWuVTroMXQ/7mbzHtdtPRa2gwDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"48fbab24233900e2b2452b456592d1116b33945c07b59d0b700a214af0a92230","last_reissued_at":"2026-07-05T10:26:50.469666Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:26:50.469666Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Is Your Video Language Model a Reliable Judge?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Ming Liu, Wensheng Zhang","submitted_at":"2025-03-07T23:17:59Z","abstract_excerpt":"As video language models (VLMs) gain more applications in various scenarios, the need for robust and scalable evaluation of their performance becomes increasingly critical. The traditional human expert-based evaluation of VLMs has limitations in consistency and scalability, which sparked interest in automatic methods such as employing VLMs to evaluate VLMs. However, the reliability of VLMs as judges remains underexplored. Existing methods often rely on a single VLM as the evaluator. However, this approach can be unreliable or biased because such a model may lack the ability to fully understand"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.05977","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.05977/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.05977","created_at":"2026-07-05T10:26:50.469725+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.05977v1","created_at":"2026-07-05T10:26:50.469725+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.05977","created_at":"2026-07-05T10:26:50.469725+00:00"},{"alias_kind":"pith_short_12","alias_value":"JD52WJBDHEAO","created_at":"2026-07-05T10:26:50.469725+00:00"},{"alias_kind":"pith_short_16","alias_value":"JD52WJBDHEAOFMSF","created_at":"2026-07-05T10:26:50.469725+00:00"},{"alias_kind":"pith_short_8","alias_value":"JD52WJBD","created_at":"2026-07-05T10:26:50.469725+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.28757","citing_title":"A Physics-Grounded Benchmark for Multi-Agent Dynamics in World Models","ref_index":38,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JD52WJBDHEAOFMSFFNCWLEWRCF","json":"https://pith.science/pith/JD52WJBDHEAOFMSFFNCWLEWRCF.json","graph_json":"https://pith.science/api/pith-number/JD52WJBDHEAOFMSFFNCWLEWRCF/graph.json","events_json":"https://pith.science/api/pith-number/JD52WJBDHEAOFMSFFNCWLEWRCF/events.json","paper":"https://pith.science/paper/JD52WJBD"},"agent_actions":{"view_html":"https://pith.science/pith/JD52WJBDHEAOFMSFFNCWLEWRCF","download_json":"https://pith.science/pith/JD52WJBDHEAOFMSFFNCWLEWRCF.json","view_paper":"https://pith.science/paper/JD52WJBD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.05977&json=true","fetch_graph":"https://pith.science/api/pith-number/JD52WJBDHEAOFMSFFNCWLEWRCF/graph.json","fetch_events":"https://pith.science/api/pith-number/JD52WJBDHEAOFMSFFNCWLEWRCF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JD52WJBDHEAOFMSFFNCWLEWRCF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JD52WJBDHEAOFMSFFNCWLEWRCF/action/storage_attestation","attest_author":"https://pith.science/pith/JD52WJBDHEAOFMSFFNCWLEWRCF/action/author_attestation","sign_citation":"https://pith.science/pith/JD52WJBDHEAOFMSFFNCWLEWRCF/action/citation_signature","submit_replication":"https://pith.science/pith/JD52WJBDHEAOFMSFFNCWLEWRCF/action/replication_record"}},"created_at":"2026-07-05T10:26:50.469725+00:00","updated_at":"2026-07-05T10:26:50.469725+00:00"}