{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:DA5WZ6BCUZLG6OIM5Q2AXAK5QW","short_pith_number":"pith:DA5WZ6BC","schema_version":"1.0","canonical_sha256":"183b6cf822a6566f390cec340b815d859b1a653ecb345e5cd80e00c072f8ac72","source":{"kind":"arxiv","id":"2505.19091","version":1},"attestation_state":"computed","paper":{"title":"ReadBench: Measuring the Dense Text Visual Reading Ability of Vision-Language Models","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.LG"],"primary_cat":"cs.CL","authors_text":"Benjamin Clavi\\'e, Florian Brand","submitted_at":"2025-05-25T11:02:01Z","abstract_excerpt":"Recent advancements in Large Vision-Language Models (VLMs), have greatly enhanced their capability to jointly process text and images. However, despite extensive benchmarks evaluating visual comprehension (e.g., diagrams, color schemes, OCR tasks...), there is limited assessment of VLMs' ability to read and reason about text-rich images effectively. To fill this gap, we introduce ReadBench, a multimodal benchmark specifically designed to evaluate the reading comprehension capabilities of VLMs. ReadBench transposes contexts from established text-only benchmarks into images of text while keeping"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.19091","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-25T11:02:01Z","cross_cats_sorted":["cs.AI","cs.CV","cs.LG"],"title_canon_sha256":"4c3d3b94bb9fae3ba51bbd36d2788459c19d9676bdf246ff0e4e3eae500f172c","abstract_canon_sha256":"d7bcb930dc5389c89b3c4370e3937deec9b196c72d12f7a20351af02209d5477"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:09:24.739219Z","signature_b64":"S2l4d+2gJP2hG7cKqM7KDnbWSo8PPJNcQrGIoZP7ODoUwEfKVylA2viOF1HwuQ9cnhDF358KKUZTOOAeQlfIBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"183b6cf822a6566f390cec340b815d859b1a653ecb345e5cd80e00c072f8ac72","last_reissued_at":"2026-07-05T11:09:24.738697Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:09:24.738697Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ReadBench: Measuring the Dense Text Visual Reading Ability of Vision-Language Models","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.LG"],"primary_cat":"cs.CL","authors_text":"Benjamin Clavi\\'e, Florian Brand","submitted_at":"2025-05-25T11:02:01Z","abstract_excerpt":"Recent advancements in Large Vision-Language Models (VLMs), have greatly enhanced their capability to jointly process text and images. However, despite extensive benchmarks evaluating visual comprehension (e.g., diagrams, color schemes, OCR tasks...), there is limited assessment of VLMs' ability to read and reason about text-rich images effectively. To fill this gap, we introduce ReadBench, a multimodal benchmark specifically designed to evaluate the reading comprehension capabilities of VLMs. ReadBench transposes contexts from established text-only benchmarks into images of text while keeping"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.19091","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.19091/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.19091","created_at":"2026-07-05T11:09:24.738763+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.19091v1","created_at":"2026-07-05T11:09:24.738763+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.19091","created_at":"2026-07-05T11:09:24.738763+00:00"},{"alias_kind":"pith_short_12","alias_value":"DA5WZ6BCUZLG","created_at":"2026-07-05T11:09:24.738763+00:00"},{"alias_kind":"pith_short_16","alias_value":"DA5WZ6BCUZLG6OIM","created_at":"2026-07-05T11:09:24.738763+00:00"},{"alias_kind":"pith_short_8","alias_value":"DA5WZ6BC","created_at":"2026-07-05T11:09:24.738763+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2510.05038","citing_title":"Guided Query Refinement: Multimodal Hybrid Retrieval with Test-Time Optimization","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DA5WZ6BCUZLG6OIM5Q2AXAK5QW","json":"https://pith.science/pith/DA5WZ6BCUZLG6OIM5Q2AXAK5QW.json","graph_json":"https://pith.science/api/pith-number/DA5WZ6BCUZLG6OIM5Q2AXAK5QW/graph.json","events_json":"https://pith.science/api/pith-number/DA5WZ6BCUZLG6OIM5Q2AXAK5QW/events.json","paper":"https://pith.science/paper/DA5WZ6BC"},"agent_actions":{"view_html":"https://pith.science/pith/DA5WZ6BCUZLG6OIM5Q2AXAK5QW","download_json":"https://pith.science/pith/DA5WZ6BCUZLG6OIM5Q2AXAK5QW.json","view_paper":"https://pith.science/paper/DA5WZ6BC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.19091&json=true","fetch_graph":"https://pith.science/api/pith-number/DA5WZ6BCUZLG6OIM5Q2AXAK5QW/graph.json","fetch_events":"https://pith.science/api/pith-number/DA5WZ6BCUZLG6OIM5Q2AXAK5QW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DA5WZ6BCUZLG6OIM5Q2AXAK5QW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DA5WZ6BCUZLG6OIM5Q2AXAK5QW/action/storage_attestation","attest_author":"https://pith.science/pith/DA5WZ6BCUZLG6OIM5Q2AXAK5QW/action/author_attestation","sign_citation":"https://pith.science/pith/DA5WZ6BCUZLG6OIM5Q2AXAK5QW/action/citation_signature","submit_replication":"https://pith.science/pith/DA5WZ6BCUZLG6OIM5Q2AXAK5QW/action/replication_record"}},"created_at":"2026-07-05T11:09:24.738763+00:00","updated_at":"2026-07-05T11:09:24.738763+00:00"}