{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:R4GTQRHD54PGYF36FLLGFIZOIK","short_pith_number":"pith:R4GTQRHD","schema_version":"1.0","canonical_sha256":"8f0d3844e3ef1e6c177e2ad662a32e42b282bbc5e6e5617847de7b2c2654c447","source":{"kind":"arxiv","id":"1909.04696","version":1},"attestation_state":"computed","paper":{"title":"Sunny and Dark Outside?! Improving Answer Consistency in VQA through Entailed Question Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Ajay Divakaran, Arijit Ray, Giedrius Burachas, Karan Sikka, Stefan Lee","submitted_at":"2019-09-10T18:18:45Z","abstract_excerpt":"While models for Visual Question Answering (VQA) have steadily improved over the years, interacting with one quickly reveals that these models lack consistency. For instance, if a model answers \"red\" to \"What color is the balloon?\", it might answer \"no\" if asked, \"Is the balloon red?\". These responses violate simple notions of entailment and raise questions about how effectively VQA models ground language. In this work, we introduce a dataset, ConVQA, and metrics that enable quantitative evaluation of consistency in VQA. For a given observable fact in an image (e.g. the balloon's color), we ge"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1909.04696","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2019-09-10T18:18:45Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"38ca3a609df31ce1da5dfb287be47e852497d9327360644a53dd907ce9326e04","abstract_canon_sha256":"d08a42a63e44d3b649032710be0fc9f2a068fd1821d34eb64b7d2472ff3e7fe8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:03:59.720801Z","signature_b64":"eHWzKD0a4zEdLwUBzi3Ejq4ZoK2krzztRsxUbhzqq6tfX/REbnPsBp9cxEoE7Go0CL2YzdE0/sQeRj6YjwzkCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8f0d3844e3ef1e6c177e2ad662a32e42b282bbc5e6e5617847de7b2c2654c447","last_reissued_at":"2026-07-05T00:03:59.720300Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:03:59.720300Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Sunny and Dark Outside?! Improving Answer Consistency in VQA through Entailed Question Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Ajay Divakaran, Arijit Ray, Giedrius Burachas, Karan Sikka, Stefan Lee","submitted_at":"2019-09-10T18:18:45Z","abstract_excerpt":"While models for Visual Question Answering (VQA) have steadily improved over the years, interacting with one quickly reveals that these models lack consistency. For instance, if a model answers \"red\" to \"What color is the balloon?\", it might answer \"no\" if asked, \"Is the balloon red?\". These responses violate simple notions of entailment and raise questions about how effectively VQA models ground language. In this work, we introduce a dataset, ConVQA, and metrics that enable quantitative evaluation of consistency in VQA. For a given observable fact in an image (e.g. the balloon's color), we ge"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1909.04696","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1909.04696/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1909.04696","created_at":"2026-07-05T00:03:59.720378+00:00"},{"alias_kind":"arxiv_version","alias_value":"1909.04696v1","created_at":"2026-07-05T00:03:59.720378+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1909.04696","created_at":"2026-07-05T00:03:59.720378+00:00"},{"alias_kind":"pith_short_12","alias_value":"R4GTQRHD54PG","created_at":"2026-07-05T00:03:59.720378+00:00"},{"alias_kind":"pith_short_16","alias_value":"R4GTQRHD54PGYF36","created_at":"2026-07-05T00:03:59.720378+00:00"},{"alias_kind":"pith_short_8","alias_value":"R4GTQRHD","created_at":"2026-07-05T00:03:59.720378+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.13875","citing_title":"Common-agency Games for Multi-Objective Test-Time Alignment","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/R4GTQRHD54PGYF36FLLGFIZOIK","json":"https://pith.science/pith/R4GTQRHD54PGYF36FLLGFIZOIK.json","graph_json":"https://pith.science/api/pith-number/R4GTQRHD54PGYF36FLLGFIZOIK/graph.json","events_json":"https://pith.science/api/pith-number/R4GTQRHD54PGYF36FLLGFIZOIK/events.json","paper":"https://pith.science/paper/R4GTQRHD"},"agent_actions":{"view_html":"https://pith.science/pith/R4GTQRHD54PGYF36FLLGFIZOIK","download_json":"https://pith.science/pith/R4GTQRHD54PGYF36FLLGFIZOIK.json","view_paper":"https://pith.science/paper/R4GTQRHD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1909.04696&json=true","fetch_graph":"https://pith.science/api/pith-number/R4GTQRHD54PGYF36FLLGFIZOIK/graph.json","fetch_events":"https://pith.science/api/pith-number/R4GTQRHD54PGYF36FLLGFIZOIK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/R4GTQRHD54PGYF36FLLGFIZOIK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/R4GTQRHD54PGYF36FLLGFIZOIK/action/storage_attestation","attest_author":"https://pith.science/pith/R4GTQRHD54PGYF36FLLGFIZOIK/action/author_attestation","sign_citation":"https://pith.science/pith/R4GTQRHD54PGYF36FLLGFIZOIK/action/citation_signature","submit_replication":"https://pith.science/pith/R4GTQRHD54PGYF36FLLGFIZOIK/action/replication_record"}},"created_at":"2026-07-05T00:03:59.720378+00:00","updated_at":"2026-07-05T00:03:59.720378+00:00"}