{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:CVOSG6RFXTQDGYLHKH3Y4LU3ZM","short_pith_number":"pith:CVOSG6RF","schema_version":"1.0","canonical_sha256":"155d237a25bce033616751f78e2e9bcb318aaa2c955155d05d95a68cd06ac1da","source":{"kind":"arxiv","id":"2402.12121","version":2},"attestation_state":"computed","paper":{"title":"IRR: Image Review Ranking Framework for Evaluating Vision-Language Models","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.MM"],"primary_cat":"cs.CL","authors_text":"Hidetaka Kamigaito, Katsuhiko Hayashi, Kazuki Hayashi, Kazuma Onishi, Seiji Gobara, Shigeki Saito, Taro Watanabe, Toma Suzuki, Yusuke Ide, Yusuke Sakai","submitted_at":"2024-02-19T13:16:10Z","abstract_excerpt":"Large-scale Vision-Language Models (LVLMs) process both images and text, excelling in multimodal tasks such as image captioning and description generation. However, while these models excel at generating factual content, their ability to generate and evaluate texts reflecting perspectives on the same image, depending on the context, has not been sufficiently explored. To address this, we propose IRR: Image Review Rank, a novel evaluation framework designed to assess critic review texts from multiple perspectives. IRR evaluates LVLMs by measuring how closely their judgments align with human int"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.12121","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-02-19T13:16:10Z","cross_cats_sorted":["cs.AI","cs.CV","cs.MM"],"title_canon_sha256":"75e6eef5e85ac8ae242703a85d0361c521359b6c3fb777b5ca61c74f59235706","abstract_canon_sha256":"60911853e06d77e2c982d4f90a6e740dc819a0b8f58cdf754e171c1ddbfc3262"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:49:41.221635Z","signature_b64":"adssAmWB6Jm5YRSoEhG7ZhHkyE9inWDArcdQ64y3YJxLQjtnvjbuRHZjTyyt78dPERjXyGuBGSmUwVz2laPIBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"155d237a25bce033616751f78e2e9bcb318aaa2c955155d05d95a68cd06ac1da","last_reissued_at":"2026-07-05T09:49:41.221108Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:49:41.221108Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"IRR: Image Review Ranking Framework for Evaluating Vision-Language Models","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.MM"],"primary_cat":"cs.CL","authors_text":"Hidetaka Kamigaito, Katsuhiko Hayashi, Kazuki Hayashi, Kazuma Onishi, Seiji Gobara, Shigeki Saito, Taro Watanabe, Toma Suzuki, Yusuke Ide, Yusuke Sakai","submitted_at":"2024-02-19T13:16:10Z","abstract_excerpt":"Large-scale Vision-Language Models (LVLMs) process both images and text, excelling in multimodal tasks such as image captioning and description generation. However, while these models excel at generating factual content, their ability to generate and evaluate texts reflecting perspectives on the same image, depending on the context, has not been sufficiently explored. To address this, we propose IRR: Image Review Rank, a novel evaluation framework designed to assess critic review texts from multiple perspectives. IRR evaluates LVLMs by measuring how closely their judgments align with human int"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.12121","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.12121/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.12121","created_at":"2026-07-05T09:49:41.221187+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.12121v2","created_at":"2026-07-05T09:49:41.221187+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.12121","created_at":"2026-07-05T09:49:41.221187+00:00"},{"alias_kind":"pith_short_12","alias_value":"CVOSG6RFXTQD","created_at":"2026-07-05T09:49:41.221187+00:00"},{"alias_kind":"pith_short_16","alias_value":"CVOSG6RFXTQDGYLH","created_at":"2026-07-05T09:49:41.221187+00:00"},{"alias_kind":"pith_short_8","alias_value":"CVOSG6RF","created_at":"2026-07-05T09:49:41.221187+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CVOSG6RFXTQDGYLHKH3Y4LU3ZM","json":"https://pith.science/pith/CVOSG6RFXTQDGYLHKH3Y4LU3ZM.json","graph_json":"https://pith.science/api/pith-number/CVOSG6RFXTQDGYLHKH3Y4LU3ZM/graph.json","events_json":"https://pith.science/api/pith-number/CVOSG6RFXTQDGYLHKH3Y4LU3ZM/events.json","paper":"https://pith.science/paper/CVOSG6RF"},"agent_actions":{"view_html":"https://pith.science/pith/CVOSG6RFXTQDGYLHKH3Y4LU3ZM","download_json":"https://pith.science/pith/CVOSG6RFXTQDGYLHKH3Y4LU3ZM.json","view_paper":"https://pith.science/paper/CVOSG6RF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.12121&json=true","fetch_graph":"https://pith.science/api/pith-number/CVOSG6RFXTQDGYLHKH3Y4LU3ZM/graph.json","fetch_events":"https://pith.science/api/pith-number/CVOSG6RFXTQDGYLHKH3Y4LU3ZM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CVOSG6RFXTQDGYLHKH3Y4LU3ZM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CVOSG6RFXTQDGYLHKH3Y4LU3ZM/action/storage_attestation","attest_author":"https://pith.science/pith/CVOSG6RFXTQDGYLHKH3Y4LU3ZM/action/author_attestation","sign_citation":"https://pith.science/pith/CVOSG6RFXTQDGYLHKH3Y4LU3ZM/action/citation_signature","submit_replication":"https://pith.science/pith/CVOSG6RFXTQDGYLHKH3Y4LU3ZM/action/replication_record"}},"created_at":"2026-07-05T09:49:41.221187+00:00","updated_at":"2026-07-05T09:49:41.221187+00:00"}