{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:KVK52U3WKROYH32OCDCOXEEZIH","short_pith_number":"pith:KVK52U3W","schema_version":"1.0","canonical_sha256":"5555dd5376545d83ef4e10c4eb909941dc7dcf2d333103339b82aad4c3c38de5","source":{"kind":"arxiv","id":"2012.02356","version":2},"attestation_state":"computed","paper":{"title":"WeaQA: Weak Supervision via Captions for Visual Question Answering","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Chitta Baral, Pratyay Banerjee, Tejas Gokhale, Yezhou Yang","submitted_at":"2020-12-04T01:22:05Z","abstract_excerpt":"Methodologies for training visual question answering (VQA) models assume the availability of datasets with human-annotated \\textit{Image-Question-Answer} (I-Q-A) triplets. This has led to heavy reliance on datasets and a lack of generalization to new types of questions and scenes. Linguistic priors along with biases and errors due to annotator subjectivity have been shown to percolate into VQA models trained on such samples. We study whether models can be trained without any human-annotated Q-A pairs, but only with images and their associated textual descriptions or captions. We present a meth"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2012.02356","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2020-12-04T01:22:05Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"c2e07f26a4dfa025c8dfe1291c70f5b4f7c82490f41828003fa821c923a21b2e","abstract_canon_sha256":"7cc093c92f9ea8cfdca45ed12e7e16f4cd5e0618bba1c27638f392db614710de"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:44:04.237027Z","signature_b64":"C8VnfagiqCUowoiRar5RL3t8M1JMwRvxOyQ1DVRO3hVYE5ffcl4PyFIX9mBuYesRosYM1E581yOcBLUt2+jBDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5555dd5376545d83ef4e10c4eb909941dc7dcf2d333103339b82aad4c3c38de5","last_reissued_at":"2026-07-05T02:44:04.236679Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:44:04.236679Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"WeaQA: Weak Supervision via Captions for Visual Question Answering","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Chitta Baral, Pratyay Banerjee, Tejas Gokhale, Yezhou Yang","submitted_at":"2020-12-04T01:22:05Z","abstract_excerpt":"Methodologies for training visual question answering (VQA) models assume the availability of datasets with human-annotated \\textit{Image-Question-Answer} (I-Q-A) triplets. This has led to heavy reliance on datasets and a lack of generalization to new types of questions and scenes. Linguistic priors along with biases and errors due to annotator subjectivity have been shown to percolate into VQA models trained on such samples. We study whether models can be trained without any human-annotated Q-A pairs, but only with images and their associated textual descriptions or captions. We present a meth"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2012.02356","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2012.02356/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2012.02356","created_at":"2026-07-05T02:44:04.236735+00:00"},{"alias_kind":"arxiv_version","alias_value":"2012.02356v2","created_at":"2026-07-05T02:44:04.236735+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2012.02356","created_at":"2026-07-05T02:44:04.236735+00:00"},{"alias_kind":"pith_short_12","alias_value":"KVK52U3WKROY","created_at":"2026-07-05T02:44:04.236735+00:00"},{"alias_kind":"pith_short_16","alias_value":"KVK52U3WKROYH32O","created_at":"2026-07-05T02:44:04.236735+00:00"},{"alias_kind":"pith_short_8","alias_value":"KVK52U3W","created_at":"2026-07-05T02:44:04.236735+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.02402","citing_title":"RG-SAN: Rule-Guided Spatial Awareness Network for End-to-End 3D Referring Expression Segmentation","ref_index":4,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KVK52U3WKROYH32OCDCOXEEZIH","json":"https://pith.science/pith/KVK52U3WKROYH32OCDCOXEEZIH.json","graph_json":"https://pith.science/api/pith-number/KVK52U3WKROYH32OCDCOXEEZIH/graph.json","events_json":"https://pith.science/api/pith-number/KVK52U3WKROYH32OCDCOXEEZIH/events.json","paper":"https://pith.science/paper/KVK52U3W"},"agent_actions":{"view_html":"https://pith.science/pith/KVK52U3WKROYH32OCDCOXEEZIH","download_json":"https://pith.science/pith/KVK52U3WKROYH32OCDCOXEEZIH.json","view_paper":"https://pith.science/paper/KVK52U3W","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2012.02356&json=true","fetch_graph":"https://pith.science/api/pith-number/KVK52U3WKROYH32OCDCOXEEZIH/graph.json","fetch_events":"https://pith.science/api/pith-number/KVK52U3WKROYH32OCDCOXEEZIH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KVK52U3WKROYH32OCDCOXEEZIH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KVK52U3WKROYH32OCDCOXEEZIH/action/storage_attestation","attest_author":"https://pith.science/pith/KVK52U3WKROYH32OCDCOXEEZIH/action/author_attestation","sign_citation":"https://pith.science/pith/KVK52U3WKROYH32OCDCOXEEZIH/action/citation_signature","submit_replication":"https://pith.science/pith/KVK52U3WKROYH32OCDCOXEEZIH/action/replication_record"}},"created_at":"2026-07-05T02:44:04.236735+00:00","updated_at":"2026-07-05T02:44:04.236735+00:00"}