{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:5ZKPT7S4OPJNBM4OYG7NOQLQC5","short_pith_number":"pith:5ZKPT7S4","schema_version":"1.0","canonical_sha256":"ee54f9fe5c73d2d0b38ec1bed74170177efe1fd93ac98d11a781d3d32d502766","source":{"kind":"arxiv","id":"2111.08896","version":3},"attestation_state":"computed","paper":{"title":"Achieving Human Parity on Visual Question Answering","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Bin Bi, Chenliang Li, Fan Wang, Fei Huang, Haiyang Xu, Ji Zhang, Junfeng Tian, Luo Si, Ming Yan, Qiyu Zhang, Rong Jin, Songfang Huang, Weihua Chen, Wei Wang, Xianzhe Xu, Zheng Cao, Zhicheng Zhang","submitted_at":"2021-11-17T04:25:11Z","abstract_excerpt":"The Visual Question Answering (VQA) task utilizes both visual image and language analysis to answer a textual question with respect to an image. It has been a popular research topic with an increasing number of real-world applications in the last decade. This paper describes our recent research of AliceMind-MMU (ALIbaba's Collection of Encoder-decoders from Machine IntelligeNce lab of Damo academy - MultiMedia Understanding) that obtains similar or even slightly better results than human being does on VQA. This is achieved by systematically improving the VQA pipeline including: (1) pre-trainin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2111.08896","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2021-11-17T04:25:11Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"6b67b930d441f887572318cc9ea56c8175e8727f79eeaa616e0dbbefdddd4500","abstract_canon_sha256":"87ff3b95504f97475f50132866b91ab6f890c62202ab403c956130467e3f2384"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:33:25.040995Z","signature_b64":"eFr/T8Fe3Ba+BY3Zjx7NiczzFTJCe7iLlgOjZXBCj+8D+XCt/3N/5CAwB+P/ozhqkf4QIPOLDz2rqy/FNUjgCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ee54f9fe5c73d2d0b38ec1bed74170177efe1fd93ac98d11a781d3d32d502766","last_reissued_at":"2026-07-05T03:33:25.037667Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:33:25.037667Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Achieving Human Parity on Visual Question Answering","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Bin Bi, Chenliang Li, Fan Wang, Fei Huang, Haiyang Xu, Ji Zhang, Junfeng Tian, Luo Si, Ming Yan, Qiyu Zhang, Rong Jin, Songfang Huang, Weihua Chen, Wei Wang, Xianzhe Xu, Zheng Cao, Zhicheng Zhang","submitted_at":"2021-11-17T04:25:11Z","abstract_excerpt":"The Visual Question Answering (VQA) task utilizes both visual image and language analysis to answer a textual question with respect to an image. It has been a popular research topic with an increasing number of real-world applications in the last decade. This paper describes our recent research of AliceMind-MMU (ALIbaba's Collection of Encoder-decoders from Machine IntelligeNce lab of Damo academy - MultiMedia Understanding) that obtains similar or even slightly better results than human being does on VQA. This is achieved by systematically improving the VQA pipeline including: (1) pre-trainin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2111.08896","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2111.08896/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2111.08896","created_at":"2026-07-05T03:33:25.038002+00:00"},{"alias_kind":"arxiv_version","alias_value":"2111.08896v3","created_at":"2026-07-05T03:33:25.038002+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2111.08896","created_at":"2026-07-05T03:33:25.038002+00:00"},{"alias_kind":"pith_short_12","alias_value":"5ZKPT7S4OPJN","created_at":"2026-07-05T03:33:25.038002+00:00"},{"alias_kind":"pith_short_16","alias_value":"5ZKPT7S4OPJNBM4O","created_at":"2026-07-05T03:33:25.038002+00:00"},{"alias_kind":"pith_short_8","alias_value":"5ZKPT7S4","created_at":"2026-07-05T03:33:25.038002+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2204.14198","citing_title":"Flamingo: a Visual Language Model for Few-Shot Learning","ref_index":134,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5ZKPT7S4OPJNBM4OYG7NOQLQC5","json":"https://pith.science/pith/5ZKPT7S4OPJNBM4OYG7NOQLQC5.json","graph_json":"https://pith.science/api/pith-number/5ZKPT7S4OPJNBM4OYG7NOQLQC5/graph.json","events_json":"https://pith.science/api/pith-number/5ZKPT7S4OPJNBM4OYG7NOQLQC5/events.json","paper":"https://pith.science/paper/5ZKPT7S4"},"agent_actions":{"view_html":"https://pith.science/pith/5ZKPT7S4OPJNBM4OYG7NOQLQC5","download_json":"https://pith.science/pith/5ZKPT7S4OPJNBM4OYG7NOQLQC5.json","view_paper":"https://pith.science/paper/5ZKPT7S4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2111.08896&json=true","fetch_graph":"https://pith.science/api/pith-number/5ZKPT7S4OPJNBM4OYG7NOQLQC5/graph.json","fetch_events":"https://pith.science/api/pith-number/5ZKPT7S4OPJNBM4OYG7NOQLQC5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5ZKPT7S4OPJNBM4OYG7NOQLQC5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5ZKPT7S4OPJNBM4OYG7NOQLQC5/action/storage_attestation","attest_author":"https://pith.science/pith/5ZKPT7S4OPJNBM4OYG7NOQLQC5/action/author_attestation","sign_citation":"https://pith.science/pith/5ZKPT7S4OPJNBM4OYG7NOQLQC5/action/citation_signature","submit_replication":"https://pith.science/pith/5ZKPT7S4OPJNBM4OYG7NOQLQC5/action/replication_record"}},"created_at":"2026-07-05T03:33:25.038002+00:00","updated_at":"2026-07-05T03:33:25.038002+00:00"}