{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:FUI45HWXIPS5YG224N2GCSQXQT","short_pith_number":"pith:FUI45HWX","schema_version":"1.0","canonical_sha256":"2d11ce9ed743e5dc1b5ae374614a1784e1a3b031fea85d5008c9dfbda7438f4c","source":{"kind":"arxiv","id":"2101.11272","version":2},"attestation_state":"computed","paper":{"title":"VisualMRC: Machine Reading Comprehension on Document Images","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Kyosuke Nishida, Ryota Tanaka, Sen Yoshida","submitted_at":"2021-01-27T09:03:06Z","abstract_excerpt":"Recent studies on machine reading comprehension have focused on text-level understanding but have not yet reached the level of human understanding of the visual layout and content of real-world documents. In this study, we introduce a new visual machine reading comprehension dataset, named VisualMRC, wherein given a question and a document image, a machine reads and comprehends texts in the image to answer the question in natural language. Compared with existing visual question answering (VQA) datasets that contain texts in images, VisualMRC focuses more on developing natural language understa"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2101.11272","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-01-27T09:03:06Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"160626d0036443f46485ad6b0c421d6b7a8c7b1f2648ec19ecc466e501a8b1e7","abstract_canon_sha256":"306a0ccf8d616a80fb8a64750c69864ad76e8b8887eac755c9c8909d5190b19d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:38:32.634289Z","signature_b64":"gZ6UWJa+mFJwBXwcE5zbz9JKNfNsM/OmvUrnMA6E9qzM29A64c3ZCEcq/bKENwjL/0UjMeEHEwEtV3miaMPUBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2d11ce9ed743e5dc1b5ae374614a1784e1a3b031fea85d5008c9dfbda7438f4c","last_reissued_at":"2026-07-05T02:38:32.633811Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:38:32.633811Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VisualMRC: Machine Reading Comprehension on Document Images","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Kyosuke Nishida, Ryota Tanaka, Sen Yoshida","submitted_at":"2021-01-27T09:03:06Z","abstract_excerpt":"Recent studies on machine reading comprehension have focused on text-level understanding but have not yet reached the level of human understanding of the visual layout and content of real-world documents. In this study, we introduce a new visual machine reading comprehension dataset, named VisualMRC, wherein given a question and a document image, a machine reads and comprehends texts in the image to answer the question in natural language. Compared with existing visual question answering (VQA) datasets that contain texts in images, VisualMRC focuses more on developing natural language understa"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2101.11272","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2101.11272/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2101.11272","created_at":"2026-07-05T02:38:32.633877+00:00"},{"alias_kind":"arxiv_version","alias_value":"2101.11272v2","created_at":"2026-07-05T02:38:32.633877+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2101.11272","created_at":"2026-07-05T02:38:32.633877+00:00"},{"alias_kind":"pith_short_12","alias_value":"FUI45HWXIPS5","created_at":"2026-07-05T02:38:32.633877+00:00"},{"alias_kind":"pith_short_16","alias_value":"FUI45HWXIPS5YG22","created_at":"2026-07-05T02:38:32.633877+00:00"},{"alias_kind":"pith_short_8","alias_value":"FUI45HWX","created_at":"2026-07-05T02:38:32.633877+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2501.02235","citing_title":"Survey on Question Answering over Visually Rich Documents: Methods, Challenges, and Trends","ref_index":88,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FUI45HWXIPS5YG224N2GCSQXQT","json":"https://pith.science/pith/FUI45HWXIPS5YG224N2GCSQXQT.json","graph_json":"https://pith.science/api/pith-number/FUI45HWXIPS5YG224N2GCSQXQT/graph.json","events_json":"https://pith.science/api/pith-number/FUI45HWXIPS5YG224N2GCSQXQT/events.json","paper":"https://pith.science/paper/FUI45HWX"},"agent_actions":{"view_html":"https://pith.science/pith/FUI45HWXIPS5YG224N2GCSQXQT","download_json":"https://pith.science/pith/FUI45HWXIPS5YG224N2GCSQXQT.json","view_paper":"https://pith.science/paper/FUI45HWX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2101.11272&json=true","fetch_graph":"https://pith.science/api/pith-number/FUI45HWXIPS5YG224N2GCSQXQT/graph.json","fetch_events":"https://pith.science/api/pith-number/FUI45HWXIPS5YG224N2GCSQXQT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FUI45HWXIPS5YG224N2GCSQXQT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FUI45HWXIPS5YG224N2GCSQXQT/action/storage_attestation","attest_author":"https://pith.science/pith/FUI45HWXIPS5YG224N2GCSQXQT/action/author_attestation","sign_citation":"https://pith.science/pith/FUI45HWXIPS5YG224N2GCSQXQT/action/citation_signature","submit_replication":"https://pith.science/pith/FUI45HWXIPS5YG224N2GCSQXQT/action/replication_record"}},"created_at":"2026-07-05T02:38:32.633877+00:00","updated_at":"2026-07-05T02:38:32.633877+00:00"}