{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:CY5XWTUFJXFZ2LGSGGLMMNKXKY","short_pith_number":"pith:CY5XWTUF","schema_version":"1.0","canonical_sha256":"163b7b4e854dcb9d2cd23196c63557561f8714a4ac1b5ac0a79a5db587b5c4e2","source":{"kind":"arxiv","id":"2308.06595","version":4},"attestation_state":"computed","paper":{"title":"VisIT-Bench: A Benchmark for Vision-Language Instruction Following Inspired by Real-World Use","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.CL","authors_text":"Anas Awadalla, Hritik Bansal, Jack Hessel, Josh Gardner, Ludwig Schmidt, Rohan Taori, Rulin Shao, Wanrong Zhu, Yonatan Bitton","submitted_at":"2023-08-12T15:27:51Z","abstract_excerpt":"We introduce VisIT-Bench (Visual InsTruction Benchmark), a benchmark for evaluation of instruction-following vision-language models for real-world use. Our starting point is curating 70 'instruction families' that we envision instruction tuned vision-language models should be able to address. Extending beyond evaluations like VQAv2 and COCO, tasks range from basic recognition to game playing and creative generation. Following curation, our dataset comprises 592 test queries, each with a human-authored instruction-conditioned caption. These descriptions surface instruction-specific factors, e.g"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.06595","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-08-12T15:27:51Z","cross_cats_sorted":["cs.AI","cs.CV"],"title_canon_sha256":"fd8e60d12c04651f2319efefd7b6620dc6b88111c0b36cf13591a4ee4912b9c0","abstract_canon_sha256":"69930e99031ad49ec4d617f2c9c4b607fc43fd377b8524f0d0b3374c17a162ff"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:27:49.038026Z","signature_b64":"/pPurhJ+lmAHHk4CnQ5G1Paig906SrlP4iyx23S/+EH7/qNNBQAzobMGig7n/Spe9Ed1meGvPsVWu5Ryw5iBBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"163b7b4e854dcb9d2cd23196c63557561f8714a4ac1b5ac0a79a5db587b5c4e2","last_reissued_at":"2026-07-05T07:27:49.037535Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:27:49.037535Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VisIT-Bench: A Benchmark for Vision-Language Instruction Following Inspired by Real-World Use","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV"],"primary_cat":"cs.CL","authors_text":"Anas Awadalla, Hritik Bansal, Jack Hessel, Josh Gardner, Ludwig Schmidt, Rohan Taori, Rulin Shao, Wanrong Zhu, Yonatan Bitton","submitted_at":"2023-08-12T15:27:51Z","abstract_excerpt":"We introduce VisIT-Bench (Visual InsTruction Benchmark), a benchmark for evaluation of instruction-following vision-language models for real-world use. Our starting point is curating 70 'instruction families' that we envision instruction tuned vision-language models should be able to address. Extending beyond evaluations like VQAv2 and COCO, tasks range from basic recognition to game playing and creative generation. Following curation, our dataset comprises 592 test queries, each with a human-authored instruction-conditioned caption. These descriptions surface instruction-specific factors, e.g"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.06595","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.06595/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.06595","created_at":"2026-07-05T07:27:49.037593+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.06595v4","created_at":"2026-07-05T07:27:49.037593+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.06595","created_at":"2026-07-05T07:27:49.037593+00:00"},{"alias_kind":"pith_short_12","alias_value":"CY5XWTUFJXFZ","created_at":"2026-07-05T07:27:49.037593+00:00"},{"alias_kind":"pith_short_16","alias_value":"CY5XWTUFJXFZ2LGS","created_at":"2026-07-05T07:27:49.037593+00:00"},{"alias_kind":"pith_short_8","alias_value":"CY5XWTUF","created_at":"2026-07-05T07:27:49.037593+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22476","citing_title":"CVSBench: A Comprehensive Benchmark for Cross-view Spatial Reasoning and Dreaming","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06217","citing_title":"DisasterBench: A Multimodal Benchmark for UAV-Based Disaster Response in Complex Environments","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30556","citing_title":"Poller: Are LLMs Suitable for Evaluating the Poetry Understanding Task?","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2405.19088","citing_title":"Cracking the Code of Juxtaposition: Can AI Models Understand the Humorous Contradictions","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2410.14702","citing_title":"Polymath: A Challenging Multi-modal Mathematical Reasoning Benchmark","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2503.23137","citing_title":"When 'YES' Meets 'BUT': Can Large Models Comprehend Contradictory Humor Through Comparative Reasoning?","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2404.13076","citing_title":"LLM Evaluators Recognize and Favor Their Own Generations","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2406.03520","citing_title":"VideoPhy: Evaluating Physical Commonsense for Video Generation","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2408.13257","citing_title":"MME-RealWorld: Could Your Multimodal LLM Challenge High-Resolution Real-World Scenarios that are Difficult for Humans?","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18803","citing_title":"LLM-as-Judge Framework for Evaluating Tone-Induced Hallucination in Vision-Language Models","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CY5XWTUFJXFZ2LGSGGLMMNKXKY","json":"https://pith.science/pith/CY5XWTUFJXFZ2LGSGGLMMNKXKY.json","graph_json":"https://pith.science/api/pith-number/CY5XWTUFJXFZ2LGSGGLMMNKXKY/graph.json","events_json":"https://pith.science/api/pith-number/CY5XWTUFJXFZ2LGSGGLMMNKXKY/events.json","paper":"https://pith.science/paper/CY5XWTUF"},"agent_actions":{"view_html":"https://pith.science/pith/CY5XWTUFJXFZ2LGSGGLMMNKXKY","download_json":"https://pith.science/pith/CY5XWTUFJXFZ2LGSGGLMMNKXKY.json","view_paper":"https://pith.science/paper/CY5XWTUF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.06595&json=true","fetch_graph":"https://pith.science/api/pith-number/CY5XWTUFJXFZ2LGSGGLMMNKXKY/graph.json","fetch_events":"https://pith.science/api/pith-number/CY5XWTUFJXFZ2LGSGGLMMNKXKY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CY5XWTUFJXFZ2LGSGGLMMNKXKY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CY5XWTUFJXFZ2LGSGGLMMNKXKY/action/storage_attestation","attest_author":"https://pith.science/pith/CY5XWTUFJXFZ2LGSGGLMMNKXKY/action/author_attestation","sign_citation":"https://pith.science/pith/CY5XWTUFJXFZ2LGSGGLMMNKXKY/action/citation_signature","submit_replication":"https://pith.science/pith/CY5XWTUFJXFZ2LGSGGLMMNKXKY/action/replication_record"}},"created_at":"2026-07-05T07:27:49.037593+00:00","updated_at":"2026-07-05T07:27:49.037593+00:00"}