{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:RPVVU7CMBRXAGHP7DULNXCMJTL","short_pith_number":"pith:RPVVU7CM","schema_version":"1.0","canonical_sha256":"8beb5a7c4c0c6e031dff1d16db89899add5e3235dee1f06de8a0b60e52cf692d","source":{"kind":"arxiv","id":"2409.00106","version":1},"attestation_state":"computed","paper":{"title":"Zero-Shot Visual Reasoning by Vision-Language Models: Benchmarking and Analysis","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.LG"],"primary_cat":"cs.CL","authors_text":"Aishik Nagar, Cheston Tan, Shantanu Jaiswal","submitted_at":"2024-08-27T14:43:54Z","abstract_excerpt":"Vision-language models (VLMs) have shown impressive zero- and few-shot performance on real-world visual question answering (VQA) benchmarks, alluding to their capabilities as visual reasoning engines. However, the benchmarks being used conflate \"pure\" visual reasoning with world knowledge, and also have questions that involve a limited number of reasoning steps. Thus, it remains unclear whether a VLM's apparent visual reasoning performance is due to its world knowledge, or due to actual visual reasoning capabilities.\n  To clarify this ambiguity, we systematically benchmark and dissect the zero"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.00106","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-08-27T14:43:54Z","cross_cats_sorted":["cs.AI","cs.CV","cs.LG"],"title_canon_sha256":"ec836e7462c7b545cc4cb98daa0c33c23d1888177999d502d21d03839a3ccfb2","abstract_canon_sha256":"0738c4db8049fa97b435755699ad25cb575461f4e725135a3489b9df9bd032ce"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:01:40.989189Z","signature_b64":"AxSAV3SCr2gxnSpGBNLsmX7goocuHnuNAd1XlfwW4Nw0ZslB8v60nUMjsSH0A8vCggCbvmrjAWF3zLYKCYGNAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8beb5a7c4c0c6e031dff1d16db89899add5e3235dee1f06de8a0b60e52cf692d","last_reissued_at":"2026-07-05T09:01:40.988747Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:01:40.988747Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Zero-Shot Visual Reasoning by Vision-Language Models: Benchmarking and Analysis","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.LG"],"primary_cat":"cs.CL","authors_text":"Aishik Nagar, Cheston Tan, Shantanu Jaiswal","submitted_at":"2024-08-27T14:43:54Z","abstract_excerpt":"Vision-language models (VLMs) have shown impressive zero- and few-shot performance on real-world visual question answering (VQA) benchmarks, alluding to their capabilities as visual reasoning engines. However, the benchmarks being used conflate \"pure\" visual reasoning with world knowledge, and also have questions that involve a limited number of reasoning steps. Thus, it remains unclear whether a VLM's apparent visual reasoning performance is due to its world knowledge, or due to actual visual reasoning capabilities.\n  To clarify this ambiguity, we systematically benchmark and dissect the zero"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.00106","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.00106/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.00106","created_at":"2026-07-05T09:01:40.988799+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.00106v1","created_at":"2026-07-05T09:01:40.988799+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.00106","created_at":"2026-07-05T09:01:40.988799+00:00"},{"alias_kind":"pith_short_12","alias_value":"RPVVU7CMBRXA","created_at":"2026-07-05T09:01:40.988799+00:00"},{"alias_kind":"pith_short_16","alias_value":"RPVVU7CMBRXAGHP7","created_at":"2026-07-05T09:01:40.988799+00:00"},{"alias_kind":"pith_short_8","alias_value":"RPVVU7CM","created_at":"2026-07-05T09:01:40.988799+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.17077","citing_title":"SubstationAI: Multimodal Large Model-Based Approaches for Analyzing Substation Equipment Faults","ref_index":35,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RPVVU7CMBRXAGHP7DULNXCMJTL","json":"https://pith.science/pith/RPVVU7CMBRXAGHP7DULNXCMJTL.json","graph_json":"https://pith.science/api/pith-number/RPVVU7CMBRXAGHP7DULNXCMJTL/graph.json","events_json":"https://pith.science/api/pith-number/RPVVU7CMBRXAGHP7DULNXCMJTL/events.json","paper":"https://pith.science/paper/RPVVU7CM"},"agent_actions":{"view_html":"https://pith.science/pith/RPVVU7CMBRXAGHP7DULNXCMJTL","download_json":"https://pith.science/pith/RPVVU7CMBRXAGHP7DULNXCMJTL.json","view_paper":"https://pith.science/paper/RPVVU7CM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.00106&json=true","fetch_graph":"https://pith.science/api/pith-number/RPVVU7CMBRXAGHP7DULNXCMJTL/graph.json","fetch_events":"https://pith.science/api/pith-number/RPVVU7CMBRXAGHP7DULNXCMJTL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RPVVU7CMBRXAGHP7DULNXCMJTL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RPVVU7CMBRXAGHP7DULNXCMJTL/action/storage_attestation","attest_author":"https://pith.science/pith/RPVVU7CMBRXAGHP7DULNXCMJTL/action/author_attestation","sign_citation":"https://pith.science/pith/RPVVU7CMBRXAGHP7DULNXCMJTL/action/citation_signature","submit_replication":"https://pith.science/pith/RPVVU7CMBRXAGHP7DULNXCMJTL/action/replication_record"}},"created_at":"2026-07-05T09:01:40.988799+00:00","updated_at":"2026-07-05T09:01:40.988799+00:00"}