{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:S2PTKWEKFZ72HLWORUXLOZ6X5Q","short_pith_number":"pith:S2PTKWEK","schema_version":"1.0","canonical_sha256":"969f35588a2e7fa3aece8d2eb767d7ec188c378c50183f764205dcb925bdd4c3","source":{"kind":"arxiv","id":"2401.06591","version":1},"attestation_state":"computed","paper":{"title":"Prometheus-Vision: Vision-Language Model as a Judge for Fine-Grained Evaluation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Geewook Kim, Minjoon Seo, Seongyun Lee, Seungone Kim, Sue Hyun Park","submitted_at":"2024-01-12T14:19:23Z","abstract_excerpt":"Assessing long-form responses generated by Vision-Language Models (VLMs) is challenging. It not only requires checking whether the VLM follows the given instruction but also verifying whether the text output is properly grounded on the given image. Inspired by the recent approach of evaluating LMs with LMs, in this work, we propose to evaluate VLMs with VLMs. For this purpose, we present a new feedback dataset called the Perception Collection, encompassing 15K customized score rubrics that users might care about during assessment. Using the Perception Collection, we train Prometheus-Vision, th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.06591","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-01-12T14:19:23Z","cross_cats_sorted":[],"title_canon_sha256":"7e33c0ecd1ada2ed63beaac77b0097c7b8441e38bfe2137dca88a1f2fd58aee7","abstract_canon_sha256":"b43c1f40e757c6c23933e5a2987c26f1e0f9dde7fa6f7160443489bf7b9b0fe1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:33:03.208296Z","signature_b64":"AcR7xJNhwvGDfQdZTPL26gea7dT+Ni+tqHuoSyHiRvlIxIYqtvY6MIbIz45gGYSBC6Josq8LKQCKH6DyrUaXBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"969f35588a2e7fa3aece8d2eb767d7ec188c378c50183f764205dcb925bdd4c3","last_reissued_at":"2026-07-05T07:33:03.207859Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:33:03.207859Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Prometheus-Vision: Vision-Language Model as a Judge for Fine-Grained Evaluation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Geewook Kim, Minjoon Seo, Seongyun Lee, Seungone Kim, Sue Hyun Park","submitted_at":"2024-01-12T14:19:23Z","abstract_excerpt":"Assessing long-form responses generated by Vision-Language Models (VLMs) is challenging. It not only requires checking whether the VLM follows the given instruction but also verifying whether the text output is properly grounded on the given image. Inspired by the recent approach of evaluating LMs with LMs, in this work, we propose to evaluate VLMs with VLMs. For this purpose, we present a new feedback dataset called the Perception Collection, encompassing 15K customized score rubrics that users might care about during assessment. Using the Perception Collection, we train Prometheus-Vision, th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.06591","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.06591/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.06591","created_at":"2026-07-05T07:33:03.207916+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.06591v1","created_at":"2026-07-05T07:33:03.207916+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.06591","created_at":"2026-07-05T07:33:03.207916+00:00"},{"alias_kind":"pith_short_12","alias_value":"S2PTKWEKFZ72","created_at":"2026-07-05T07:33:03.207916+00:00"},{"alias_kind":"pith_short_16","alias_value":"S2PTKWEKFZ72HLWO","created_at":"2026-07-05T07:33:03.207916+00:00"},{"alias_kind":"pith_short_8","alias_value":"S2PTKWEK","created_at":"2026-07-05T07:33:03.207916+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05391","citing_title":"LLM-as-a-Verifier: A General-Purpose Verification Framework","ref_index":71,"is_internal_anchor":true},{"citing_arxiv_id":"2606.04773","citing_title":"NextMotionQA: Benchmarking and Judging Human Motion Understanding with Vision-Language Models","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27931","citing_title":"DiagramRAG: A Lightweight Framework to Retrieve Scientific Diagram for Figure Generation","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2501.05465","citing_title":"Small Language Models (SLMs) Can Still Pack a Punch: A survey (updated 2026)","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2412.05579","citing_title":"LLMs-as-Judges: A Comprehensive Survey on LLM-based Evaluation Methods","ref_index":127,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/S2PTKWEKFZ72HLWORUXLOZ6X5Q","json":"https://pith.science/pith/S2PTKWEKFZ72HLWORUXLOZ6X5Q.json","graph_json":"https://pith.science/api/pith-number/S2PTKWEKFZ72HLWORUXLOZ6X5Q/graph.json","events_json":"https://pith.science/api/pith-number/S2PTKWEKFZ72HLWORUXLOZ6X5Q/events.json","paper":"https://pith.science/paper/S2PTKWEK"},"agent_actions":{"view_html":"https://pith.science/pith/S2PTKWEKFZ72HLWORUXLOZ6X5Q","download_json":"https://pith.science/pith/S2PTKWEKFZ72HLWORUXLOZ6X5Q.json","view_paper":"https://pith.science/paper/S2PTKWEK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.06591&json=true","fetch_graph":"https://pith.science/api/pith-number/S2PTKWEKFZ72HLWORUXLOZ6X5Q/graph.json","fetch_events":"https://pith.science/api/pith-number/S2PTKWEKFZ72HLWORUXLOZ6X5Q/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/S2PTKWEKFZ72HLWORUXLOZ6X5Q/action/timestamp_anchor","attest_storage":"https://pith.science/pith/S2PTKWEKFZ72HLWORUXLOZ6X5Q/action/storage_attestation","attest_author":"https://pith.science/pith/S2PTKWEKFZ72HLWORUXLOZ6X5Q/action/author_attestation","sign_citation":"https://pith.science/pith/S2PTKWEKFZ72HLWORUXLOZ6X5Q/action/citation_signature","submit_replication":"https://pith.science/pith/S2PTKWEKFZ72HLWORUXLOZ6X5Q/action/replication_record"}},"created_at":"2026-07-05T07:33:03.207916+00:00","updated_at":"2026-07-05T07:33:03.207916+00:00"}