{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2017:IMGFGIOFMHK7P4NFWVHWHYSU4Y","short_pith_number":"pith:IMGFGIOF","schema_version":"1.0","canonical_sha256":"430c5321c561d5f7f1a5b54f63e254e61c62ced409c84554cee01d9799a66a4d","source":{"kind":"arxiv","id":"1703.09684","version":2},"attestation_state":"computed","paper":{"title":"An Analysis of Visual Question Answering Algorithms","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Christopher Kanan, Kushal Kafle","submitted_at":"2017-03-28T17:48:07Z","abstract_excerpt":"In visual question answering (VQA), an algorithm must answer text-based questions about images. While multiple datasets for VQA have been created since late 2014, they all have flaws in both their content and the way algorithms are evaluated on them. As a result, evaluation scores are inflated and predominantly determined by answering easier questions, making it difficult to compare different methods. In this paper, we analyze existing VQA algorithms using a new dataset. It contains over 1.6 million questions organized into 12 different categories. We also introduce questions that are meaningl"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1703.09684","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2017-03-28T17:48:07Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"29ceb629dd78b21571e5a7ccd3d83652f7bee5118815815ebe6c87af60370b98","abstract_canon_sha256":"c1b63d7a6cc11202027698ce6832361c857b0df8c65ec4e79a34a50941c584f1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-18T00:35:13.760442Z","signature_b64":"Zz7N4X3r769ci65/FIS/v6XpAqS2+FjoLKYrtjh4YFm6sM7poCs0OCxxdwl770cvx4m2U+WMZ1IiaRPvzUTPDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"430c5321c561d5f7f1a5b54f63e254e61c62ced409c84554cee01d9799a66a4d","last_reissued_at":"2026-05-18T00:35:13.759952Z","signature_status":"signed_v1","first_computed_at":"2026-05-18T00:35:13.759952Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"An Analysis of Visual Question Answering Algorithms","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Christopher Kanan, Kushal Kafle","submitted_at":"2017-03-28T17:48:07Z","abstract_excerpt":"In visual question answering (VQA), an algorithm must answer text-based questions about images. While multiple datasets for VQA have been created since late 2014, they all have flaws in both their content and the way algorithms are evaluated on them. As a result, evaluation scores are inflated and predominantly determined by answering easier questions, making it difficult to compare different methods. In this paper, we analyze existing VQA algorithms using a new dataset. It contains over 1.6 million questions organized into 12 different categories. We also introduce questions that are meaningl"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1703.09684","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1703.09684","created_at":"2026-05-18T00:35:13.760038+00:00"},{"alias_kind":"arxiv_version","alias_value":"1703.09684v2","created_at":"2026-05-18T00:35:13.760038+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1703.09684","created_at":"2026-05-18T00:35:13.760038+00:00"},{"alias_kind":"pith_short_12","alias_value":"IMGFGIOFMHK7","created_at":"2026-05-18T12:31:21.493067+00:00"},{"alias_kind":"pith_short_16","alias_value":"IMGFGIOFMHK7P4NF","created_at":"2026-05-18T12:31:21.493067+00:00"},{"alias_kind":"pith_short_8","alias_value":"IMGFGIOF","created_at":"2026-05-18T12:31:21.493067+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2501.07109","citing_title":"The Quest for Visual Understanding: A Journey Through the Evolution of Visual Question Answering","ref_index":38,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IMGFGIOFMHK7P4NFWVHWHYSU4Y","json":"https://pith.science/pith/IMGFGIOFMHK7P4NFWVHWHYSU4Y.json","graph_json":"https://pith.science/api/pith-number/IMGFGIOFMHK7P4NFWVHWHYSU4Y/graph.json","events_json":"https://pith.science/api/pith-number/IMGFGIOFMHK7P4NFWVHWHYSU4Y/events.json","paper":"https://pith.science/paper/IMGFGIOF"},"agent_actions":{"view_html":"https://pith.science/pith/IMGFGIOFMHK7P4NFWVHWHYSU4Y","download_json":"https://pith.science/pith/IMGFGIOFMHK7P4NFWVHWHYSU4Y.json","view_paper":"https://pith.science/paper/IMGFGIOF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1703.09684&json=true","fetch_graph":"https://pith.science/api/pith-number/IMGFGIOFMHK7P4NFWVHWHYSU4Y/graph.json","fetch_events":"https://pith.science/api/pith-number/IMGFGIOFMHK7P4NFWVHWHYSU4Y/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IMGFGIOFMHK7P4NFWVHWHYSU4Y/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IMGFGIOFMHK7P4NFWVHWHYSU4Y/action/storage_attestation","attest_author":"https://pith.science/pith/IMGFGIOFMHK7P4NFWVHWHYSU4Y/action/author_attestation","sign_citation":"https://pith.science/pith/IMGFGIOFMHK7P4NFWVHWHYSU4Y/action/citation_signature","submit_replication":"https://pith.science/pith/IMGFGIOFMHK7P4NFWVHWHYSU4Y/action/replication_record"}},"created_at":"2026-05-18T00:35:13.760038+00:00","updated_at":"2026-05-18T00:35:13.760038+00:00"}