{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:SNVMO3T4AIDJF77LGVH5YC2PMT","short_pith_number":"pith:SNVMO3T4","schema_version":"1.0","canonical_sha256":"936ac76e7c020692ffeb354fdc0b4f64fe4dce12a1e3bdaabd23ba9dd9f70a91","source":{"kind":"arxiv","id":"2309.16511","version":1},"attestation_state":"computed","paper":{"title":"Toloka Visual Question Answering Benchmark","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.HC"],"primary_cat":"cs.CV","authors_text":"Alisa Smirnova, Daniil Likhobaba, Dmitry Ustalov, Nikita Pavlichenko, Sergey Koshelev","submitted_at":"2023-09-28T15:18:35Z","abstract_excerpt":"In this paper, we present Toloka Visual Question Answering, a new crowdsourced dataset allowing comparing performance of machine learning systems against human level of expertise in the grounding visual question answering task. In this task, given an image and a textual question, one has to draw the bounding box around the object correctly responding to that question. Every image-question pair contains the response, with only one correct response per image. Our dataset contains 45,199 pairs of images and questions in English, provided with ground truth bounding boxes, split into train and two "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.16511","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-09-28T15:18:35Z","cross_cats_sorted":["cs.AI","cs.CL","cs.HC"],"title_canon_sha256":"cc977a35a293fe022e077f3e6bf1281e8d6093ae7dd0a5670ae3cdbd40904c46","abstract_canon_sha256":"8c618ee8b31c2adea1cb809922cd06e751daa79a71229e33e3f7266be1849ba1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:55:17.679370Z","signature_b64":"G2Nx83uwZqtb11wyO/zXDP1DWGM5jdSbWifaUn2Cd/gqjqDymCTo7OAD6t1Fuo71zPgAH7NtSqJhGjBpVabjAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"936ac76e7c020692ffeb354fdc0b4f64fe4dce12a1e3bdaabd23ba9dd9f70a91","last_reissued_at":"2026-07-05T06:55:17.678819Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:55:17.678819Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Toloka Visual Question Answering Benchmark","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.HC"],"primary_cat":"cs.CV","authors_text":"Alisa Smirnova, Daniil Likhobaba, Dmitry Ustalov, Nikita Pavlichenko, Sergey Koshelev","submitted_at":"2023-09-28T15:18:35Z","abstract_excerpt":"In this paper, we present Toloka Visual Question Answering, a new crowdsourced dataset allowing comparing performance of machine learning systems against human level of expertise in the grounding visual question answering task. In this task, given an image and a textual question, one has to draw the bounding box around the object correctly responding to that question. Every image-question pair contains the response, with only one correct response per image. Our dataset contains 45,199 pairs of images and questions in English, provided with ground truth bounding boxes, split into train and two "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.16511","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.16511/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.16511","created_at":"2026-07-05T06:55:17.678884+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.16511v1","created_at":"2026-07-05T06:55:17.678884+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.16511","created_at":"2026-07-05T06:55:17.678884+00:00"},{"alias_kind":"pith_short_12","alias_value":"SNVMO3T4AIDJ","created_at":"2026-07-05T06:55:17.678884+00:00"},{"alias_kind":"pith_short_16","alias_value":"SNVMO3T4AIDJF77L","created_at":"2026-07-05T06:55:17.678884+00:00"},{"alias_kind":"pith_short_8","alias_value":"SNVMO3T4","created_at":"2026-07-05T06:55:17.678884+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.10651","citing_title":"Kwai Keye-VL-2.0 Technical Report","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28551","citing_title":"DataComp-VLM: Improved Open Datasets for Vision-Language Models","ref_index":289,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28551","citing_title":"DataComp-VLM: Improved Open Datasets for Vision-Language Models","ref_index":289,"is_internal_anchor":false},{"citing_arxiv_id":"2312.14238","citing_title":"InternVL: Scaling up Vision Foundation Models and Aligning for Generic Visual-Linguistic Tasks","ref_index":141,"is_internal_anchor":false},{"citing_arxiv_id":"2412.05271","citing_title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","ref_index":237,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SNVMO3T4AIDJF77LGVH5YC2PMT","json":"https://pith.science/pith/SNVMO3T4AIDJF77LGVH5YC2PMT.json","graph_json":"https://pith.science/api/pith-number/SNVMO3T4AIDJF77LGVH5YC2PMT/graph.json","events_json":"https://pith.science/api/pith-number/SNVMO3T4AIDJF77LGVH5YC2PMT/events.json","paper":"https://pith.science/paper/SNVMO3T4"},"agent_actions":{"view_html":"https://pith.science/pith/SNVMO3T4AIDJF77LGVH5YC2PMT","download_json":"https://pith.science/pith/SNVMO3T4AIDJF77LGVH5YC2PMT.json","view_paper":"https://pith.science/paper/SNVMO3T4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.16511&json=true","fetch_graph":"https://pith.science/api/pith-number/SNVMO3T4AIDJF77LGVH5YC2PMT/graph.json","fetch_events":"https://pith.science/api/pith-number/SNVMO3T4AIDJF77LGVH5YC2PMT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SNVMO3T4AIDJF77LGVH5YC2PMT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SNVMO3T4AIDJF77LGVH5YC2PMT/action/storage_attestation","attest_author":"https://pith.science/pith/SNVMO3T4AIDJF77LGVH5YC2PMT/action/author_attestation","sign_citation":"https://pith.science/pith/SNVMO3T4AIDJF77LGVH5YC2PMT/action/citation_signature","submit_replication":"https://pith.science/pith/SNVMO3T4AIDJF77LGVH5YC2PMT/action/replication_record"}},"created_at":"2026-07-05T06:55:17.678884+00:00","updated_at":"2026-07-05T06:55:17.678884+00:00"}