{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:2LYVHEY76XU3QOVRSSDWJ545KY","short_pith_number":"pith:2LYVHEY7","schema_version":"1.0","canonical_sha256":"d2f153931ff5e9b83ab1948764f79d5624c612d04b92c103e0519078da906c59","source":{"kind":"arxiv","id":"2504.13945","version":4},"attestation_state":"computed","paper":{"title":"Evaluating Menu OCR and Translation: A Benchmark for Aligning Human and Automated Evaluations in Large Vision-Language Models","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Chong Li, Hao Yang, Junhao Zhu, Mengli Zhu, Ning Xie, Pengfei Li, Shiliang Sun, Shuang Wu, Tengfei Song, Weidong Zhang, Zhanglin Wu","submitted_at":"2025-04-16T03:08:57Z","abstract_excerpt":"The rapid advancement of large vision-language models (LVLMs) has significantly propelled applications in document understanding, particularly in optical character recognition (OCR) and multilingual translation. However, current evaluations of LVLMs, like the widely used OCRBench, mainly focus on verifying the correctness of their short-text responses and long-text responses with simple layout, while the evaluation of their ability to understand long texts with complex layout design is highly significant but largely overlooked. In this paper, we propose Menu OCR and Translation Benchmark (MOTB"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.13945","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2025-04-16T03:08:57Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"0fbbf9878243c2877ccc9b76d00bd52043f3e97cf70041b0c59f3c3e8bc21696","abstract_canon_sha256":"bd4cd600f0b558432f864d81bebd41d72eb6e8ddd414801f6aeba8b25b129d95"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:04:49.285474Z","signature_b64":"CSl+fQ52x9DPOxEAf9rKMdazTx4ZnqYDrbgiagXPukGJim2nui3SvQQQKSFOOccAxN/nJzY6Q/ZKWYz36rrFAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d2f153931ff5e9b83ab1948764f79d5624c612d04b92c103e0519078da906c59","last_reissued_at":"2026-07-05T11:04:49.284922Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:04:49.284922Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluating Menu OCR and Translation: A Benchmark for Aligning Human and Automated Evaluations in Large Vision-Language Models","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Chong Li, Hao Yang, Junhao Zhu, Mengli Zhu, Ning Xie, Pengfei Li, Shiliang Sun, Shuang Wu, Tengfei Song, Weidong Zhang, Zhanglin Wu","submitted_at":"2025-04-16T03:08:57Z","abstract_excerpt":"The rapid advancement of large vision-language models (LVLMs) has significantly propelled applications in document understanding, particularly in optical character recognition (OCR) and multilingual translation. However, current evaluations of LVLMs, like the widely used OCRBench, mainly focus on verifying the correctness of their short-text responses and long-text responses with simple layout, while the evaluation of their ability to understand long texts with complex layout design is highly significant but largely overlooked. In this paper, we propose Menu OCR and Translation Benchmark (MOTB"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.13945","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.13945/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.13945","created_at":"2026-07-05T11:04:49.284995+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.13945v4","created_at":"2026-07-05T11:04:49.284995+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.13945","created_at":"2026-07-05T11:04:49.284995+00:00"},{"alias_kind":"pith_short_12","alias_value":"2LYVHEY76XU3","created_at":"2026-07-05T11:04:49.284995+00:00"},{"alias_kind":"pith_short_16","alias_value":"2LYVHEY76XU3QOVR","created_at":"2026-07-05T11:04:49.284995+00:00"},{"alias_kind":"pith_short_8","alias_value":"2LYVHEY7","created_at":"2026-07-05T11:04:49.284995+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2504.17315","citing_title":"DIMT25@ICDAR2025: HW-TSC's End-to-End Document Image Machine Translation System Leveraging Large Vision-Language Model","ref_index":27,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2LYVHEY76XU3QOVRSSDWJ545KY","json":"https://pith.science/pith/2LYVHEY76XU3QOVRSSDWJ545KY.json","graph_json":"https://pith.science/api/pith-number/2LYVHEY76XU3QOVRSSDWJ545KY/graph.json","events_json":"https://pith.science/api/pith-number/2LYVHEY76XU3QOVRSSDWJ545KY/events.json","paper":"https://pith.science/paper/2LYVHEY7"},"agent_actions":{"view_html":"https://pith.science/pith/2LYVHEY76XU3QOVRSSDWJ545KY","download_json":"https://pith.science/pith/2LYVHEY76XU3QOVRSSDWJ545KY.json","view_paper":"https://pith.science/paper/2LYVHEY7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.13945&json=true","fetch_graph":"https://pith.science/api/pith-number/2LYVHEY76XU3QOVRSSDWJ545KY/graph.json","fetch_events":"https://pith.science/api/pith-number/2LYVHEY76XU3QOVRSSDWJ545KY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2LYVHEY76XU3QOVRSSDWJ545KY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2LYVHEY76XU3QOVRSSDWJ545KY/action/storage_attestation","attest_author":"https://pith.science/pith/2LYVHEY76XU3QOVRSSDWJ545KY/action/author_attestation","sign_citation":"https://pith.science/pith/2LYVHEY76XU3QOVRSSDWJ545KY/action/citation_signature","submit_replication":"https://pith.science/pith/2LYVHEY76XU3QOVRSSDWJ545KY/action/replication_record"}},"created_at":"2026-07-05T11:04:49.284995+00:00","updated_at":"2026-07-05T11:04:49.284995+00:00"}