{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:ZMZBLCC4S3ADLZZFFFDFK4AKWG","short_pith_number":"pith:ZMZBLCC4","schema_version":"1.0","canonical_sha256":"cb3215885c96c035e725294655700ab18f27fb1186bb4d8e025a3d07bf28bca1","source":{"kind":"arxiv","id":"2203.07613","version":1},"attestation_state":"computed","paper":{"title":"CARETS: A Consistency And Robustness Evaluative Test Suite for VQA","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Carlos E. Jimenez, Karthik Narasimhan, Olga Russakovsky","submitted_at":"2022-03-15T03:01:03Z","abstract_excerpt":"We introduce CARETS, a systematic test suite to measure consistency and robustness of modern VQA models through a series of six fine-grained capability tests. In contrast to existing VQA test sets, CARETS features balanced question generation to create pairs of instances to test models, with each pair focusing on a specific capability such as rephrasing, logical symmetry or image obfuscation. We evaluate six modern VQA systems on CARETS and identify several actionable weaknesses in model comprehension, especially with concepts such as negation, disjunction, or hypernym invariance. Interestingl"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2203.07613","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2022-03-15T03:01:03Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"e70845545d3e469f764f5b3a25bcc41f5f1d15462ab87d7157932218146b1700","abstract_canon_sha256":"a3370630fb2bf1b5ce6c832dfa3cc09e589fa1adb81cc2e88dc20b3f5a3cff16"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:05:20.550906Z","signature_b64":"4gto0fySXYaENJffLuVvcz4mNdIUwr9mQNVPVAXsOjcSsr2jrB4aD+XeIBi/Ic4iaQe2ZKnwIu2tD+cV8c8ODA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cb3215885c96c035e725294655700ab18f27fb1186bb4d8e025a3d07bf28bca1","last_reissued_at":"2026-07-05T04:05:20.550527Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:05:20.550527Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CARETS: A Consistency And Robustness Evaluative Test Suite for VQA","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Carlos E. Jimenez, Karthik Narasimhan, Olga Russakovsky","submitted_at":"2022-03-15T03:01:03Z","abstract_excerpt":"We introduce CARETS, a systematic test suite to measure consistency and robustness of modern VQA models through a series of six fine-grained capability tests. In contrast to existing VQA test sets, CARETS features balanced question generation to create pairs of instances to test models, with each pair focusing on a specific capability such as rephrasing, logical symmetry or image obfuscation. We evaluate six modern VQA systems on CARETS and identify several actionable weaknesses in model comprehension, especially with concepts such as negation, disjunction, or hypernym invariance. Interestingl"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2203.07613","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2203.07613/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2203.07613","created_at":"2026-07-05T04:05:20.550581+00:00"},{"alias_kind":"arxiv_version","alias_value":"2203.07613v1","created_at":"2026-07-05T04:05:20.550581+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2203.07613","created_at":"2026-07-05T04:05:20.550581+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZMZBLCC4S3AD","created_at":"2026-07-05T04:05:20.550581+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZMZBLCC4S3ADLZZF","created_at":"2026-07-05T04:05:20.550581+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZMZBLCC4","created_at":"2026-07-05T04:05:20.550581+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.14672","citing_title":"FiVL: A Framework for Improved Vision-Language Alignment through the Lens of Training, Evaluation and Explainability","ref_index":13,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZMZBLCC4S3ADLZZFFFDFK4AKWG","json":"https://pith.science/pith/ZMZBLCC4S3ADLZZFFFDFK4AKWG.json","graph_json":"https://pith.science/api/pith-number/ZMZBLCC4S3ADLZZFFFDFK4AKWG/graph.json","events_json":"https://pith.science/api/pith-number/ZMZBLCC4S3ADLZZFFFDFK4AKWG/events.json","paper":"https://pith.science/paper/ZMZBLCC4"},"agent_actions":{"view_html":"https://pith.science/pith/ZMZBLCC4S3ADLZZFFFDFK4AKWG","download_json":"https://pith.science/pith/ZMZBLCC4S3ADLZZFFFDFK4AKWG.json","view_paper":"https://pith.science/paper/ZMZBLCC4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2203.07613&json=true","fetch_graph":"https://pith.science/api/pith-number/ZMZBLCC4S3ADLZZFFFDFK4AKWG/graph.json","fetch_events":"https://pith.science/api/pith-number/ZMZBLCC4S3ADLZZFFFDFK4AKWG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZMZBLCC4S3ADLZZFFFDFK4AKWG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZMZBLCC4S3ADLZZFFFDFK4AKWG/action/storage_attestation","attest_author":"https://pith.science/pith/ZMZBLCC4S3ADLZZFFFDFK4AKWG/action/author_attestation","sign_citation":"https://pith.science/pith/ZMZBLCC4S3ADLZZFFFDFK4AKWG/action/citation_signature","submit_replication":"https://pith.science/pith/ZMZBLCC4S3ADLZZFFFDFK4AKWG/action/replication_record"}},"created_at":"2026-07-05T04:05:20.550581+00:00","updated_at":"2026-07-05T04:05:20.550581+00:00"}