{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:UOJKSSPHIHA5OCNCBQYZTN7MM5","short_pith_number":"pith:UOJKSSPH","schema_version":"1.0","canonical_sha256":"a392a949e741c1d709a20c3199b7ec676b46d955d8f0d5340083339625ae3c4c","source":{"kind":"arxiv","id":"2405.11985","version":5},"attestation_state":"computed","paper":{"title":"MTVQA: Benchmarking Multilingual Text-Centric Visual Question Answering","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Can Huang, Chunhui Lin, Hao Feng, Hao Liu, Jinghui Lu, Jingqun Tang, Kuan Lu, Mohamad Fitri Faiz Bin Mahmood, Qi Liu, Shu Wei, Wanqing Li, Xiang Bai, Yangfan He, Yanjie Wang, YongJie Ye, Yuliang Liu, Zhen Zhao","submitted_at":"2024-05-20T12:35:01Z","abstract_excerpt":"Text-Centric Visual Question Answering (TEC-VQA) in its proper format not only facilitates human-machine interaction in text-centric visual environments but also serves as a de facto gold proxy to evaluate AI models in the domain of text-centric scene understanding. Nonetheless, most existing TEC-VQA benchmarks have focused on high-resource languages like English and Chinese. Despite pioneering works to expand multilingual QA pairs in non-text-centric VQA datasets through translation engines, the translation-based protocol encounters a substantial \"visual-textual misalignment\" problem when app"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.11985","kind":"arxiv","version":5},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-05-20T12:35:01Z","cross_cats_sorted":[],"title_canon_sha256":"777a20dbf4da8a4029a695ecff7601ceb5879082c8de6736d58ffdc42e02595d","abstract_canon_sha256":"24a3a16be9605a5c5e8ac0542e1b2eebf4b0ce7d2f4c46da3fe27b56c31c9806"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:19:19.429687Z","signature_b64":"ioo2DTmBoBAbOJfCpmvhtPJYDhBmdW3EhYFeDvht5FqxS/zNERHGo67PxP2MWqEY0lLBsH6agF0ZqQu6LwafAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a392a949e741c1d709a20c3199b7ec676b46d955d8f0d5340083339625ae3c4c","last_reissued_at":"2026-07-05T11:19:19.429171Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:19:19.429171Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MTVQA: Benchmarking Multilingual Text-Centric Visual Question Answering","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Can Huang, Chunhui Lin, Hao Feng, Hao Liu, Jinghui Lu, Jingqun Tang, Kuan Lu, Mohamad Fitri Faiz Bin Mahmood, Qi Liu, Shu Wei, Wanqing Li, Xiang Bai, Yangfan He, Yanjie Wang, YongJie Ye, Yuliang Liu, Zhen Zhao","submitted_at":"2024-05-20T12:35:01Z","abstract_excerpt":"Text-Centric Visual Question Answering (TEC-VQA) in its proper format not only facilitates human-machine interaction in text-centric visual environments but also serves as a de facto gold proxy to evaluate AI models in the domain of text-centric scene understanding. Nonetheless, most existing TEC-VQA benchmarks have focused on high-resource languages like English and Chinese. Despite pioneering works to expand multilingual QA pairs in non-text-centric VQA datasets through translation engines, the translation-based protocol encounters a substantial \"visual-textual misalignment\" problem when app"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.11985","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.11985/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.11985","created_at":"2026-07-05T11:19:19.429244+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.11985v5","created_at":"2026-07-05T11:19:19.429244+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.11985","created_at":"2026-07-05T11:19:19.429244+00:00"},{"alias_kind":"pith_short_12","alias_value":"UOJKSSPHIHA5","created_at":"2026-07-05T11:19:19.429244+00:00"},{"alias_kind":"pith_short_16","alias_value":"UOJKSSPHIHA5OCNC","created_at":"2026-07-05T11:19:19.429244+00:00"},{"alias_kind":"pith_short_8","alias_value":"UOJKSSPH","created_at":"2026-07-05T11:19:19.429244+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2604.18347","citing_title":"Multilingual Training and Evaluation Resources for Vision-Language Models","ref_index":39,"is_internal_anchor":true},{"citing_arxiv_id":"2606.30189","citing_title":"DAIN: Dynamic Agent-Based Interaction Network for Efficient and Collaborative Multimodal Reasoning","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2502.13923","citing_title":"Qwen2.5-VL Technical Report","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18173","citing_title":"Do You Need Text Rectification? Soft Attention Mask Embedding for Rectification-Free Scene Text Spotting","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2509.10026","citing_title":"LaV-CoT: Language-Aware Visual CoT with Multi-Aspect Reward Optimization for Real-World Multilingual VQA","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2509.22123","citing_title":"Multilingual Vision-Language Models, A Survey","ref_index":138,"is_internal_anchor":false},{"citing_arxiv_id":"2502.04326","citing_title":"WorldSense: Evaluating Real-world Omnimodal Understanding for Multimodal LLMs","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2408.13257","citing_title":"MME-RealWorld: Could Your Multimodal LLM Challenge High-Resolution Real-World Scenarios that are Difficult for Humans?","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03339","citing_title":"Hierarchical Awareness Adapters with Hybrid Pyramid Feature Fusion for Dense Depth Prediction","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2504.10479","citing_title":"InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models","ref_index":114,"is_internal_anchor":false},{"citing_arxiv_id":"2412.05271","citing_title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","ref_index":228,"is_internal_anchor":false},{"citing_arxiv_id":"2508.18265","citing_title":"InternVL3.5: Advancing Open-Source Multimodal Models in Versatility, Reasoning, and Efficiency","ref_index":119,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18347","citing_title":"Multilingual Training and Evaluation Resources for Vision-Language Models","ref_index":39,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UOJKSSPHIHA5OCNCBQYZTN7MM5","json":"https://pith.science/pith/UOJKSSPHIHA5OCNCBQYZTN7MM5.json","graph_json":"https://pith.science/api/pith-number/UOJKSSPHIHA5OCNCBQYZTN7MM5/graph.json","events_json":"https://pith.science/api/pith-number/UOJKSSPHIHA5OCNCBQYZTN7MM5/events.json","paper":"https://pith.science/paper/UOJKSSPH"},"agent_actions":{"view_html":"https://pith.science/pith/UOJKSSPHIHA5OCNCBQYZTN7MM5","download_json":"https://pith.science/pith/UOJKSSPHIHA5OCNCBQYZTN7MM5.json","view_paper":"https://pith.science/paper/UOJKSSPH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.11985&json=true","fetch_graph":"https://pith.science/api/pith-number/UOJKSSPHIHA5OCNCBQYZTN7MM5/graph.json","fetch_events":"https://pith.science/api/pith-number/UOJKSSPHIHA5OCNCBQYZTN7MM5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UOJKSSPHIHA5OCNCBQYZTN7MM5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UOJKSSPHIHA5OCNCBQYZTN7MM5/action/storage_attestation","attest_author":"https://pith.science/pith/UOJKSSPHIHA5OCNCBQYZTN7MM5/action/author_attestation","sign_citation":"https://pith.science/pith/UOJKSSPHIHA5OCNCBQYZTN7MM5/action/citation_signature","submit_replication":"https://pith.science/pith/UOJKSSPHIHA5OCNCBQYZTN7MM5/action/replication_record"}},"created_at":"2026-07-05T11:19:19.429244+00:00","updated_at":"2026-07-05T11:19:19.429244+00:00"}