{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:2XQHQCVVBYXP6T73LRZ7D5RSXA","short_pith_number":"pith:2XQHQCVV","schema_version":"1.0","canonical_sha256":"d5e0780ab50e2eff4ffb5c73f1f632b805c67f499fd8c88ff8e7b9db8aeef44a","source":{"kind":"arxiv","id":"2507.13405","version":1},"attestation_state":"computed","paper":{"title":"COREVQA: A Crowd Observation and Reasoning Entailment Visual Question Answering Benchmark","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Andrew Lin, Charles Duong, Hannah You, Ishant Chintapatla, Kazuma Choji, Kevin Zhu, Naaisha Agarwal, Sean O'Brien, Vasu Sharma","submitted_at":"2025-07-17T04:47:47Z","abstract_excerpt":"Recently, many benchmarks and datasets have been developed to evaluate Vision-Language Models (VLMs) using visual question answering (VQA) pairs, and models have shown significant accuracy improvements. However, these benchmarks rarely test the model's ability to accurately complete visual entailment, for instance, accepting or refuting a hypothesis based on the image. To address this, we propose COREVQA (Crowd Observations and Reasoning Entailment), a benchmark of 5608 image and synthetically generated true/false statement pairs, with images derived from the CrowdHuman dataset, to provoke vis"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.13405","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-07-17T04:47:47Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"f00fada9af72a98925fb9e65adc786e3531cdba0ff291e7f9afcc0a6a7ebc4b8","abstract_canon_sha256":"2b9ebbe6dd7941d229fe4aa00beca34ce2a132ab58d8d9701dc5fa06fbafdece"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:39:16.988533Z","signature_b64":"5R0Qfd4I2YmQT9OKoHzuMGcHJ07IbZMWUIkaaJ56atPOdDNoRtBIuDZpSlEQJ/JhuT3K+e2KzCw0lBLUGq1LBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d5e0780ab50e2eff4ffb5c73f1f632b805c67f499fd8c88ff8e7b9db8aeef44a","last_reissued_at":"2026-07-05T11:39:16.988015Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:39:16.988015Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"COREVQA: A Crowd Observation and Reasoning Entailment Visual Question Answering Benchmark","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Andrew Lin, Charles Duong, Hannah You, Ishant Chintapatla, Kazuma Choji, Kevin Zhu, Naaisha Agarwal, Sean O'Brien, Vasu Sharma","submitted_at":"2025-07-17T04:47:47Z","abstract_excerpt":"Recently, many benchmarks and datasets have been developed to evaluate Vision-Language Models (VLMs) using visual question answering (VQA) pairs, and models have shown significant accuracy improvements. However, these benchmarks rarely test the model's ability to accurately complete visual entailment, for instance, accepting or refuting a hypothesis based on the image. To address this, we propose COREVQA (Crowd Observations and Reasoning Entailment), a benchmark of 5608 image and synthetically generated true/false statement pairs, with images derived from the CrowdHuman dataset, to provoke vis"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.13405","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.13405/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.13405","created_at":"2026-07-05T11:39:16.988075+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.13405v1","created_at":"2026-07-05T11:39:16.988075+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.13405","created_at":"2026-07-05T11:39:16.988075+00:00"},{"alias_kind":"pith_short_12","alias_value":"2XQHQCVVBYXP","created_at":"2026-07-05T11:39:16.988075+00:00"},{"alias_kind":"pith_short_16","alias_value":"2XQHQCVVBYXP6T73","created_at":"2026-07-05T11:39:16.988075+00:00"},{"alias_kind":"pith_short_8","alias_value":"2XQHQCVV","created_at":"2026-07-05T11:39:16.988075+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.26601","citing_title":"FTibSuite: A Comprehensive Resource Suite for Tibetan Vision-Language Modeling","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2XQHQCVVBYXP6T73LRZ7D5RSXA","json":"https://pith.science/pith/2XQHQCVVBYXP6T73LRZ7D5RSXA.json","graph_json":"https://pith.science/api/pith-number/2XQHQCVVBYXP6T73LRZ7D5RSXA/graph.json","events_json":"https://pith.science/api/pith-number/2XQHQCVVBYXP6T73LRZ7D5RSXA/events.json","paper":"https://pith.science/paper/2XQHQCVV"},"agent_actions":{"view_html":"https://pith.science/pith/2XQHQCVVBYXP6T73LRZ7D5RSXA","download_json":"https://pith.science/pith/2XQHQCVVBYXP6T73LRZ7D5RSXA.json","view_paper":"https://pith.science/paper/2XQHQCVV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.13405&json=true","fetch_graph":"https://pith.science/api/pith-number/2XQHQCVVBYXP6T73LRZ7D5RSXA/graph.json","fetch_events":"https://pith.science/api/pith-number/2XQHQCVVBYXP6T73LRZ7D5RSXA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2XQHQCVVBYXP6T73LRZ7D5RSXA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2XQHQCVVBYXP6T73LRZ7D5RSXA/action/storage_attestation","attest_author":"https://pith.science/pith/2XQHQCVVBYXP6T73LRZ7D5RSXA/action/author_attestation","sign_citation":"https://pith.science/pith/2XQHQCVVBYXP6T73LRZ7D5RSXA/action/citation_signature","submit_replication":"https://pith.science/pith/2XQHQCVVBYXP6T73LRZ7D5RSXA/action/replication_record"}},"created_at":"2026-07-05T11:39:16.988075+00:00","updated_at":"2026-07-05T11:39:16.988075+00:00"}