{"as_of":"2026-08-07T01:01:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:65aeccb87adff4294206afc4fa593bed3a6167fd42a9acf85917956ed8352ef1","coverage":[{"denominator":122,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-17T09:55:35.452649Z","state":"measured"},{"denominator":145,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":145,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":45,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":45,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T23:51:35.327728Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-01T22:26:17.944984Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":"2305.07895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-07-01T22:26:17.944984Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","venue":"cs.CV","work_id":"521163e9-4c77-4c73-8b7f-9a6aa75b6d3b","year":2023},"citing_paper":{"arxiv_id":"2311.16502","last_updated":"2024-06-13T15:02:39Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-11-27T17:33:21Z","title":"MMMU: A Massive Multi-discipline Multimodal Understanding and Reasoning Benchmark for Expert AGI","version":4},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-15T05:37:41.401736Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2311.16502"},"observation_digest":"sha256:d0b762ca3af027cd81d8f88d63ab4dfd7877741eccbd543fda72c789b5f9a89c","observation_id":"42cb7730-504b-4137-b553-df17066e2fce","resolution":{"observed_at":"2026-05-17T09:55:36.035986Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":"2305.07895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-07-01T22:26:17.944984Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","venue":"cs.CV","work_id":"521163e9-4c77-4c73-8b7f-9a6aa75b6d3b","year":2023},"citing_paper":{"arxiv_id":"2404.12390","last_updated":"2024-07-03T08:44:45Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-04-18T17:59:54Z","title":"BLINK: Multimodal Large Language Models Can See but Not Perceive","version":4},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-15T20:18:15.439163Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2404.12390"},"observation_digest":"sha256:01d52061c7fad4605a0c9057c19bec383c1016f2084bae131ee1df7ffeb1b279","observation_id":"84113f97-4b89-46be-8fbb-382d41d8eb83","resolution":{"observed_at":"2026-05-17T09:55:36.035986Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":"2305.07895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-07-01T22:26:17.944984Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","venue":"cs.CV","work_id":"521163e9-4c77-4c73-8b7f-9a6aa75b6d3b","year":2023},"citing_paper":{"arxiv_id":"2404.16821","last_updated":"2024-04-29T20:24:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-04-25T17:59:19Z","title":"How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites","version":2},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-05-12T20:58:58.849040Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2404.16821"},"observation_digest":"sha256:6fcb5bb770fad8c5e7bf03237dc12e5413dc43c40bc868709f56417e76c45824","observation_id":"acdbea90-6207-4367-9e84-fb0544ee8b9c","resolution":{"observed_at":"2026-05-17T09:55:36.035986Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":"2305.07895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-07-01T22:26:17.944984Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","venue":"cs.CV","work_id":"521163e9-4c77-4c73-8b7f-9a6aa75b6d3b","year":2023},"citing_paper":{"arxiv_id":"2406.09411","last_updated":"2024-07-02T01:56:14Z","snapshot_observed_at":"2026-08-06T10:53:26.005728Z","submitted_at":"2024-06-13T17:59:52Z","title":"MuirBench: A Comprehensive Benchmark for Robust Multi-image Understanding","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-17T01:09:30.360275Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2406.09411"},"observation_digest":"sha256:08410ef46ae553bc7ce721d8c13c3eb53acd2d2a3dcb40f161b2152b9c1ddeec","observation_id":"caad77f5-aee2-4b81-84bb-a43d49c62b75","resolution":{"observed_at":"2026-05-17T09:55:36.035986Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":"2305.07895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-07-01T22:26:17.944984Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","venue":"cs.CV","work_id":"521163e9-4c77-4c73-8b7f-9a6aa75b6d3b","year":2023},"citing_paper":{"arxiv_id":"2406.16860","last_updated":"2024-12-04T17:57:32Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-24T17:59:42Z","title":"Cambrian-1: A Fully Open, Vision-Centric Exploration of Multimodal LLMs","version":2},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-05-17T00:05:03.547664Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2406.16860"},"observation_digest":"sha256:f1be29e48de97a33bc66cfeef38c659e26945fd618a8d8d61cde067f1ee77262","observation_id":"e6f39914-70fa-44f3-8342-15b2f7b38646","resolution":{"observed_at":"2026-05-17T09:55:36.035986Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":"2305.07895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-07-01T22:26:17.944984Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","venue":"cs.CV","work_id":"521163e9-4c77-4c73-8b7f-9a6aa75b6d3b","year":2023},"citing_paper":{"arxiv_id":"2408.01800","last_updated":"2024-08-03T15:02:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-03T15:02:21Z","title":"MiniCPM-V: A GPT-4V Level MLLM on Your Phone","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-10T21:07:31.387726Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2408.01800"},"observation_digest":"sha256:8a949ca0999a7b50b7728f320f6dab1da56bc42fa66dde2544734e36b03ecb48","observation_id":"108de9e8-c92b-41a0-9cc7-4afe1ab1f30f","resolution":{"observed_at":"2026-05-17T09:55:36.035986Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":"2305.07895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-07-01T22:26:17.944984Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","venue":"cs.CV","work_id":"521163e9-4c77-4c73-8b7f-9a6aa75b6d3b","year":2023},"citing_paper":{"arxiv_id":"2409.18839","last_updated":"2024-09-27T15:35:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-27T15:35:15Z","title":"MinerU: An Open-Source Solution for Precise Document Content Extraction","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-16T04:00:25.624430Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2409.18839"},"observation_digest":"sha256:3b3c94643353aabb05728d7769c86c5eed8dee3c15a044f866a69c9fc426c838","observation_id":"573c2021-cbf8-4bff-8a90-dd159ef5f695","resolution":{"observed_at":"2026-05-17T09:55:36.035986Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":"2305.07895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-07-01T22:26:17.944984Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","venue":"cs.CV","work_id":"521163e9-4c77-4c73-8b7f-9a6aa75b6d3b","year":2023},"citing_paper":{"arxiv_id":"2409.18869","last_updated":"2024-09-27T16:06:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-27T16:06:11Z","title":"Emu3: Next-Token Prediction is All You Need","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-05-11T10:56:06.418360Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2409.18869"},"observation_digest":"sha256:787afbd4006f529a7a0cf0947285d2819ff33ed701e6a91499c1308b21a9cd0b","observation_id":"67dfba78-7812-482c-80a9-22be0c02224b","resolution":{"observed_at":"2026-05-17T09:55:36.035986Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":"2305.07895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-07-01T22:26:17.944984Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","venue":"cs.CV","work_id":"521163e9-4c77-4c73-8b7f-9a6aa75b6d3b","year":2023},"citing_paper":{"arxiv_id":"2412.05271","last_updated":"2025-09-26T12:52:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-06T18:57:08Z","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","version":5},"reference_index":158,"source":"pdf_text","source_observed_at":"2026-05-10T13:23:57.588851Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2412.05271"},"observation_digest":"sha256:4a863426032ca178e57301c00e18d1192edba20b055df14b0fbf6adbe40a7e19","observation_id":"208d770c-af15-46fc-b797-6f53bc4d3b36","resolution":{"observed_at":"2026-05-17T09:55:36.035986Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":"2305.07895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-07-01T22:26:17.944984Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","venue":"cs.CV","work_id":"521163e9-4c77-4c73-8b7f-9a6aa75b6d3b","year":2023},"citing_paper":{"arxiv_id":"2412.10302","last_updated":"2024-12-13T17:37:48Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-13T17:37:48Z","title":"DeepSeek-VL2: Mixture-of-Experts Vision-Language Models for Advanced Multimodal Understanding","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-05-11T10:09:21.542356Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2412.10302"},"observation_digest":"sha256:0a9488becd730346f0c9882d407903650ce1ff3ca1546886f30e6ab06c513be2","observation_id":"e6c40b26-3e3a-4417-b165-85d1ec3b0867","resolution":{"observed_at":"2026-05-17T09:55:36.035986Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":"2305.07895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-07-01T22:26:17.944984Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","venue":"cs.CV","work_id":"521163e9-4c77-4c73-8b7f-9a6aa75b6d3b","year":2023},"citing_paper":{"arxiv_id":"2412.14164","last_updated":"2024-12-18T18:58:50Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-18T18:58:50Z","title":"MetaMorph: Multimodal Understanding and Generation via Instruction Tuning","version":1},"reference_index":215,"source":"arxiv_source","source_observed_at":"2026-05-17T07:51:12.953777Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2412.14164"},"observation_digest":"sha256:f37cf8ae72106ee12843f0973c4da6c1d9cab7de37a8609aeb6f89db58de5ac3","observation_id":"8ec462d2-90d2-47f9-8f25-74b21d9fd086","resolution":{"observed_at":"2026-05-17T09:55:36.035986Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":"2305.07895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-07-01T22:26:17.944984Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","venue":"cs.CV","work_id":"521163e9-4c77-4c73-8b7f-9a6aa75b6d3b","year":2023},"citing_paper":{"arxiv_id":"2501.00321","last_updated":"2025-06-05T02:59:05Z","snapshot_observed_at":"2026-08-02T18:54:27.250149Z","submitted_at":"2024-12-31T07:32:35Z","title":"OCRBench v2: An Improved Benchmark for Evaluating Large Multimodal Models on Visual Text Localization and Reasoning","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-17T20:33:26.613927Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2501.00321"},"observation_digest":"sha256:94314b0a470a62c9c3d102ee793a037c65bae0428c31e8d707e8211bba4aa583","observation_id":"1f421b4f-c3aa-4521-9c22-dc66af286c46","resolution":{"observed_at":"2026-05-17T20:33:26.701581Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":"2305.07895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-07-01T22:26:17.944984Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","venue":"cs.CV","work_id":"521163e9-4c77-4c73-8b7f-9a6aa75b6d3b","year":2023},"citing_paper":{"arxiv_id":"2501.01957","last_updated":"2025-10-24T02:32:10Z","snapshot_observed_at":"2026-08-06T10:28:03.148202Z","submitted_at":"2025-01-03T18:59:52Z","title":"VITA-1.5: Towards GPT-4o Level Real-Time Vision and Speech Interaction","version":4},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-05-17T21:08:19.570050Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2501.01957"},"observation_digest":"sha256:d7c3b39e02a5c092b4ac50f7aa71c8e7c7d7e0ac61fda3b00c2c27e150f073ee","observation_id":"3ef1718b-157d-4f35-aec1-f03a11e66af4","resolution":{"observed_at":"2026-05-17T21:08:19.754191Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":"2305.07895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-07-01T22:26:17.944984Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","venue":"cs.CV","work_id":"521163e9-4c77-4c73-8b7f-9a6aa75b6d3b","year":2023},"citing_paper":{"arxiv_id":"2501.13106","last_updated":"2025-06-03T03:33:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T18:59:46Z","title":"VideoLLaMA 3: Frontier Multimodal Foundation Models for Image and Video Understanding","version":4},"reference_index":115,"source":"pdf_text","source_observed_at":"2026-05-11T01:19:59.603343Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2501.13106"},"observation_digest":"sha256:0876712ca6098a49be6878bb02fbbd8befa078892d3d1b5f16bee2039b8f2e3c","observation_id":"0606427f-4b50-49c7-896c-983b871b52be","resolution":{"observed_at":"2026-05-17T09:55:36.035986Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":"2305.07895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-07-01T22:26:17.944984Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","venue":"cs.CV","work_id":"521163e9-4c77-4c73-8b7f-9a6aa75b6d3b","year":2023},"citing_paper":{"arxiv_id":"2502.20295","last_updated":"2026-04-18T08:27:37Z","snapshot_observed_at":"2026-08-02T21:22:35.616248Z","submitted_at":"2025-02-27T17:21:18Z","title":"Judge a Book by its Cover: Investigating Multi-Modal LLMs for Multi-Page Handwritten Document Transcription","version":2},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-05-23T02:23:22.682357Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2502.20295"},"observation_digest":"sha256:d505ff46a192e70a86f9d85b6b9298109e3e6fca1404d3aa0d761d36d4a0ad33","observation_id":"31c5c6d4-00f8-4b90-9da2-243f439cc956","resolution":{"observed_at":"2026-05-23T02:25:19.369228Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":"2305.07895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-07-01T22:26:17.944984Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","venue":"cs.CV","work_id":"521163e9-4c77-4c73-8b7f-9a6aa75b6d3b","year":2023},"citing_paper":{"arxiv_id":"2503.23733","last_updated":"2025-03-31T05:13:02Z","snapshot_observed_at":"2026-07-06T21:01:13.411586Z","submitted_at":"2025-03-31T05:13:02Z","title":"AdaMMS: Model Merging for Heterogeneous Multimodal Large Language Models with Unsupervised Coefficient Optimization","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-22T22:47:09.500229Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2503.23733"},"observation_digest":"sha256:42468ae0de277bbd8751a2ce3b9916c9f456ad6a4485be0f2360a39a597d8eae","observation_id":"cb342305-57b1-4715-8a9b-1a366456e06e","resolution":{"observed_at":"2026-05-22T22:47:13.010209Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":"2305.07895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-07-01T22:26:17.944984Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","venue":"cs.CV","work_id":"521163e9-4c77-4c73-8b7f-9a6aa75b6d3b","year":2023},"citing_paper":{"arxiv_id":"2504.10479","last_updated":"2025-04-19T03:47:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-14T17:59:25Z","title":"InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models","version":3},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-05-10T13:41:07.991012Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2504.10479"},"observation_digest":"sha256:eeaac6bc31409095b7b813c3167be7fc7709ea9b2f8582be8bf5943d5841880b","observation_id":"cbc8a068-f9e8-4fda-8e1c-d49fe18adf14","resolution":{"observed_at":"2026-05-17T09:55:36.035986Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-08-06T16:50:08.136002Z","title":"On the hidden mystery of ocr in large multimodal models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.12566","last_updated":"2025-07-16T18:31:23Z","snapshot_observed_at":"2026-08-06T16:41:50.566267Z","submitted_at":"2025-07-16T18:31:23Z","title":"Mono-InternVL-1.5: Towards Cheaper and Faster Monolithic Multimodal Large Language Models","version":1},"reference_index":126,"source":"pdf_text","source_observed_at":"2026-08-06T16:50:08.136002Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2507.12566"},"observation_digest":"sha256:8a72296264f51e1f1fe3ac34d79b27ab38a0c6168c30d64a1324f3ce16f751d1","observation_id":"5d51dff9-9bf0-4f56-b820-6832a09aad10","resolution":{"observed_at":"2026-08-06T16:50:08.136002Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-08-06T16:33:57.223241Z","title":"Ocrbench: On the hidden mystery of ocr in large multimodal models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.13348","last_updated":"2025-07-17T17:59:55Z","snapshot_observed_at":"2026-08-06T16:22:14.738964Z","submitted_at":"2025-07-17T17:59:55Z","title":"VisionThink: Smart and Efficient Vision Language Model via Reinforcement Learning","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T16:33:57.223241Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2507.13348"},"observation_digest":"sha256:0614716f347213b88838ad4ee91735ad056eb19ae7d34a290d3e245166a78e11","observation_id":"63493399-6b72-4a1e-9e33-15c4ee441e79","resolution":{"observed_at":"2026-08-06T16:33:57.223241Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-08-06T15:57:02.676446Z","title":"On the hidden mystery of ocr in large multimodal models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.14675","last_updated":"2025-07-19T16:03:34Z","snapshot_observed_at":"2026-08-06T15:47:53.180347Z","submitted_at":"2025-07-19T16:03:34Z","title":"Docopilot: Improving Multimodal Models for Document-Level Understanding","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T15:57:02.676446Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2507.14675"},"observation_digest":"sha256:1387beabcf14924f6f245dc06f9647d200ddfd4cd45788ffd4b2f6a09a2a1485","observation_id":"72e64c9b-d179-4977-9223-196cbe943d12","resolution":{"observed_at":"2026-08-06T15:57:02.676446Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-08-06T13:18:03.510688Z","title":"On the hidden mystery of OCR in large multimodal models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.20842","last_updated":"2025-07-28T13:50:53Z","snapshot_observed_at":"2026-08-06T13:17:59.155006Z","submitted_at":"2025-07-28T13:50:53Z","title":"METEOR: Multi-Encoder Collaborative Token Pruning for Efficient Vision Language Models","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T13:18:03.510688Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2507.20842"},"observation_digest":"sha256:015713a3a46a074c8f1e314d711e4456f632751e62e787b1e038ee903c59ae54","observation_id":"bce14f18-8693-40b5-a6be-ebd619080353","resolution":{"observed_at":"2026-08-06T13:18:03.510688Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":"2305.07895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-07-01T22:26:17.944984Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","venue":"cs.CV","work_id":"521163e9-4c77-4c73-8b7f-9a6aa75b6d3b","year":2023},"citing_paper":{"arxiv_id":"2508.18265","last_updated":"2025-08-27T14:39:45Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-08-25T17:58:17Z","title":"InternVL3.5: Advancing Open-Source Multimodal Models in Versatility, Reasoning, and Efficiency","version":2},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-05-10T11:58:58.660564Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2508.18265"},"observation_digest":"sha256:342d4e045b6e925f68cd60fc57f6cceb890c97003e43b9446512f9748c40b140","observation_id":"104fecb8-65c1-418b-a46d-6e6666e00833","resolution":{"observed_at":"2026-05-17T09:55:36.035986Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-08-05T14:42:21.055785Z","title":"Ocrbench: On the hidden mystery of ocr in large multimodal models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.21113","last_updated":"2025-09-02T13:37:38Z","snapshot_observed_at":"2026-08-05T14:42:19.666014Z","submitted_at":"2025-08-28T17:48:19Z","title":"R-4B: Incentivizing General-Purpose Auto-Thinking Capability in MLLMs via Bi-Mode Annealing and Reinforce Learning","version":2},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-05T14:42:21.055785Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2508.21113"},"observation_digest":"sha256:afc1f2e54e28459b735b93d7b540f922601a132d3b47f036976e697116e43306","observation_id":"af9d6896-24a2-44d2-93c1-71628b5632bf","resolution":{"observed_at":"2026-08-05T14:42:21.055785Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-08-05T12:27:27.411073Z","title":"On the hidden mystery of ocr in large multimodal models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.01610","last_updated":"2025-09-01T16:43:48Z","snapshot_observed_at":"2026-08-06T04:32:51.830706Z","submitted_at":"2025-09-01T16:43:48Z","title":"Improving Large Vision and Language Models by Learning from a Panel of Peers","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-05T12:27:27.411073Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2509.01610"},"observation_digest":"sha256:efa7fd406487cbd826c3f9e627cc32216e1a0fc758d40c507640efa4959acb73","observation_id":"6e1762ba-a8fd-48c6-992c-3711ca12ccfd","resolution":{"observed_at":"2026-08-05T12:27:27.411073Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":"2305.07895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-07-01T22:26:17.944984Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","venue":"cs.CV","work_id":"521163e9-4c77-4c73-8b7f-9a6aa75b6d3b","year":2023},"citing_paper":{"arxiv_id":"2512.20856","last_updated":"2025-12-24T00:24:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-12-24T00:24:05Z","title":"NVIDIA Nemotron 3: Efficient and Open Intelligence","version":1},"reference_index":85,"source":"arxiv_source","source_observed_at":"2026-05-18T01:40:42.190369Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2512.20856"},"observation_digest":"sha256:f835d39e224f56efe6bf0a08bf02a9ff7b555203fae28b20b5f31ea42cf2c9c1","observation_id":"1b88e53c-3418-420b-b1d8-c40a7a5a57d9","resolution":{"observed_at":"2026-05-18T01:40:42.490904Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":"2305.07895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-07-01T22:26:17.944984Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","venue":"cs.CV","work_id":"521163e9-4c77-4c73-8b7f-9a6aa75b6d3b","year":2023},"citing_paper":{"arxiv_id":"2605.11301","last_updated":"2026-05-11T22:42:12Z","snapshot_observed_at":"2026-07-06T23:23:11.928199Z","submitted_at":"2026-05-11T22:42:12Z","title":"LatentRouter: Can We Choose the Right Multimodal Model Before Seeing Its Answer?","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-13T01:42:54.802658Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2605.11301"},"observation_digest":"sha256:dbfcd60b2b957b500ef6bd67ffbc7e6601dde85ee55de434da0e44e5f58abb3a","observation_id":"3d112dc8-8dd4-4e48-a3e2-a54f6ec36e5e","resolution":{"observed_at":"2026-05-17T09:55:36.035986Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":"2305.07895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-07-01T22:26:17.944984Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","venue":"cs.CV","work_id":"521163e9-4c77-4c73-8b7f-9a6aa75b6d3b","year":2023},"citing_paper":{"arxiv_id":"2605.12623","last_updated":"2026-05-21T05:33:02Z","snapshot_observed_at":"2026-08-06T19:38:37.895679Z","submitted_at":"2026-05-12T18:09:38Z","title":"DocAtlas: Multilingual Document Understanding Across 80+ Languages","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-05-14T21:02:45.148167Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2605.12623"},"observation_digest":"sha256:ee9ed9146dffc671d2338b1c3ef59b38167e87697e959cb571658802325795ed","observation_id":"4dce1cc5-3251-4a46-a103-8c20c003df31","resolution":{"observed_at":"2026-05-17T09:55:36.035986Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":"2305.07895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-07-01T22:26:17.944984Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","venue":"cs.CV","work_id":"521163e9-4c77-4c73-8b7f-9a6aa75b6d3b","year":2023},"citing_paper":{"arxiv_id":"2605.12623","last_updated":"2026-05-21T05:33:02Z","snapshot_observed_at":"2026-08-06T19:38:37.895679Z","submitted_at":"2026-05-12T18:09:38Z","title":"DocAtlas: Multilingual Document Understanding Across 80+ Languages","version":2},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-05-22T09:51:40.160096Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2605.12623"},"observation_digest":"sha256:bd66f4672117963e6de0277ad3160f54fed69d927be516a1ce223c76ed8ca7cf","observation_id":"fb23a86e-5a27-45cd-b988-3435387d857d","resolution":{"observed_at":"2026-05-22T09:54:47.193159Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":"2305.07895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-07-01T22:26:17.944984Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","venue":"cs.CV","work_id":"521163e9-4c77-4c73-8b7f-9a6aa75b6d3b","year":2023},"citing_paper":{"arxiv_id":"2605.13080","last_updated":"2026-05-13T06:54:09Z","snapshot_observed_at":"2026-08-02T12:36:41.415898Z","submitted_at":"2026-05-13T06:54:09Z","title":"Learning to See What You Need: Gaze Attention for Multimodal Large Language Models","version":1},"reference_index":69,"source":"arxiv_source","source_observed_at":"2026-05-14T20:13:18.813131Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2605.13080"},"observation_digest":"sha256:d0b4e05721da92b5974ce47a85edcbd89c431cda978edcafc3b466028264c428","observation_id":"114da7dd-16a7-40c8-927b-6000b8750f62","resolution":{"observed_at":"2026-05-17T09:55:36.035986Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":"2305.07895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-07-01T22:26:17.944984Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","venue":"cs.CV","work_id":"521163e9-4c77-4c73-8b7f-9a6aa75b6d3b","year":2023},"citing_paper":{"arxiv_id":"2605.15876","last_updated":"2026-05-20T09:56:58Z","snapshot_observed_at":"2026-08-02T03:46:43.550240Z","submitted_at":"2026-05-15T11:54:17Z","title":"Unlocking Dense Metric Depth Estimation in VLMs","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-20T19:20:04.468206Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2605.15876"},"observation_digest":"sha256:c37331408018b6f053e176ef8084c3d93431de49d41b998d38be5b5358ca5459","observation_id":"de73b38f-659d-4630-b086-a33d87361b1a","resolution":{"observed_at":"2026-05-20T19:23:40.965067Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":"2305.07895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-07-01T22:26:17.944984Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","venue":"cs.CV","work_id":"521163e9-4c77-4c73-8b7f-9a6aa75b6d3b","year":2023},"citing_paper":{"arxiv_id":"2605.15876","last_updated":"2026-05-20T09:56:58Z","snapshot_observed_at":"2026-08-02T03:46:43.550240Z","submitted_at":"2026-05-15T11:54:17Z","title":"Unlocking Dense Metric Depth Estimation in VLMs","version":3},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-21T07:54:52.926995Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2605.15876"},"observation_digest":"sha256:3981acb5a68130359c18c7715016daae8dd4610e06c9700933c969bd2de72384","observation_id":"349e7629-183a-4fbe-a80f-f18a699563e3","resolution":{"observed_at":"2026-05-21T07:59:51.149910Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":"2305.07895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-07-01T22:26:17.944984Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","venue":"cs.CV","work_id":"521163e9-4c77-4c73-8b7f-9a6aa75b6d3b","year":2023},"citing_paper":{"arxiv_id":"2605.17159","last_updated":"2026-05-16T21:18:39Z","snapshot_observed_at":"2026-07-06T23:28:12.114488Z","submitted_at":"2026-05-16T21:18:39Z","title":"MADP: A Multi-Agent Pipeline for Sustainable Document Processing with Human-in-the-Loop","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-20T14:25:11.234515Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2605.17159"},"observation_digest":"sha256:5144882a3729ebbdc6d89e6d045010afa6855e7d0ffd21e7b3eda62d5c602eca","observation_id":"32e8d4f8-4a00-46d0-8919-eac79f87739f","resolution":{"observed_at":"2026-05-20T14:28:21.509695Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":"2305.07895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-07-01T22:26:17.944984Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","venue":"cs.CV","work_id":"521163e9-4c77-4c73-8b7f-9a6aa75b6d3b","year":2023},"citing_paper":{"arxiv_id":"2605.18359","last_updated":"2026-05-26T13:49:58Z","snapshot_observed_at":"2026-08-02T17:52:26.858372Z","submitted_at":"2026-05-18T13:12:50Z","title":"RAVE: Re-Allocating Visual Attention in Large Multimodal Models","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-20T11:10:58.225078Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2605.18359"},"observation_digest":"sha256:ee26361672eff0fbdaa1538509cdb6918885099d4d0ff661decd35995f7ccd45","observation_id":"bcfed605-6088-4464-ba83-b9db0d569cd6","resolution":{"observed_at":"2026-05-20T11:13:13.462205Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":"2305.07895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-07-01T22:26:17.944984Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","venue":"cs.CV","work_id":"521163e9-4c77-4c73-8b7f-9a6aa75b6d3b","year":2023},"citing_paper":{"arxiv_id":"2605.18359","last_updated":"2026-05-26T13:49:58Z","snapshot_observed_at":"2026-08-02T17:52:26.858372Z","submitted_at":"2026-05-18T13:12:50Z","title":"RAVE: Re-Allocating Visual Attention in Large Multimodal Models","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-30T18:36:56.248838Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2605.18359"},"observation_digest":"sha256:0fc13120d2cbd02895f7c1243c7035e8adfdf4aed09baf3d1375228801fd3ea6","observation_id":"2e6f6826-63e4-4837-8811-9a25944ae653","resolution":{"observed_at":"2026-07-01T14:55:48.338811Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":"2305.07895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-07-01T22:26:17.944984Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","venue":"cs.CV","work_id":"521163e9-4c77-4c73-8b7f-9a6aa75b6d3b","year":2023},"citing_paper":{"arxiv_id":"2605.18852","last_updated":"2026-05-13T12:18:32Z","snapshot_observed_at":"2026-08-05T13:41:13.471264Z","submitted_at":"2026-05-13T12:18:32Z","title":"Robust Checkpoint Selection for Multimodal LLMs via Agentic Evaluation and Stability-Aware Ranking","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-20T20:42:07.819116Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2605.18852"},"observation_digest":"sha256:dc8cb71f256cb92d5d7a0f029914dbe1e407e8d01795ae50eb039c0e4a1784a1","observation_id":"cf498cfb-151b-4037-adbb-69d8eae67c1b","resolution":{"observed_at":"2026-05-20T20:43:43.378259Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":"2305.07895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-07-01T22:26:17.944984Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","venue":"cs.CV","work_id":"521163e9-4c77-4c73-8b7f-9a6aa75b6d3b","year":2023},"citing_paper":{"arxiv_id":"2605.20033","last_updated":"2026-05-19T15:54:22Z","snapshot_observed_at":"2026-07-06T23:30:39.512029Z","submitted_at":"2026-05-19T15:54:22Z","title":"A Nash Equilibrium Framework For Training-Free Multimodal Step Verification","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-05-20T06:10:36.290559Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2605.20033"},"observation_digest":"sha256:64a7539cdf41d3ef4c8f8fe6ff3a114b95a49ca0a74077fd0f6edb5a29dcd5d6","observation_id":"b8fdf76b-35a3-4f86-b0bc-c8d1e014e0b0","resolution":{"observed_at":"2026-05-20T06:13:05.249487Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":"2305.07895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-07-01T22:26:17.944984Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","venue":"cs.CV","work_id":"521163e9-4c77-4c73-8b7f-9a6aa75b6d3b","year":2023},"citing_paper":{"arxiv_id":"2605.25036","last_updated":"2026-05-24T12:23:13Z","snapshot_observed_at":"2026-08-01T17:14:09.537062Z","submitted_at":"2026-05-24T12:23:13Z","title":"Language Bias in LVLMs: From In-Depth Analysis to Simple and Effective Mitigation","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-30T12:03:41.268375Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2605.25036"},"observation_digest":"sha256:8d0f3e492abe3e7451413452c5d1f91107cf8f7b328b112a4f77904dde650de0","observation_id":"e4de0cd5-b69e-4176-8f66-dc0b1fbdaefb","resolution":{"observed_at":"2026-06-30T12:04:38.653273Z","resolver_source":"local_arxiv","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":"2305.07895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-07-01T22:26:17.944984Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","venue":"cs.CV","work_id":"521163e9-4c77-4c73-8b7f-9a6aa75b6d3b","year":2023},"citing_paper":{"arxiv_id":"2606.07639","last_updated":"2026-06-01T09:07:15Z","snapshot_observed_at":"2026-07-06T23:47:15.041928Z","submitted_at":"2026-06-01T09:07:15Z","title":"MOSS-Video-Preview: Toward Real-Time Video Understanding via Cross-Attention","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-06-28T15:22:31.310003Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2606.07639"},"observation_digest":"sha256:4c6de22dab7cef5d146d561ab4112f3577852aae92a029c580a8965e7f05b9ca","observation_id":"371c0666-998b-4f2c-a4cc-9596ea918a4a","resolution":{"observed_at":"2026-07-01T22:26:17.946170Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":"2305.07895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-07-01T22:26:17.944984Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","venue":"cs.CV","work_id":"521163e9-4c77-4c73-8b7f-9a6aa75b6d3b","year":2023},"citing_paper":{"arxiv_id":"2606.24253","last_updated":"2026-06-26T09:37:54Z","snapshot_observed_at":"2026-08-06T13:21:30.445520Z","submitted_at":"2026-06-23T07:42:22Z","title":"TuringViT: Making SOTA Vision Transformers Accessible to All","version":2},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-06-29T05:32:26.746776Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2606.24253"},"observation_digest":"sha256:aa74c77f2c355890a8b42cc5a12a77bb986c1c6f11d6124b32a2bb1f7740bf8c","observation_id":"5fe77a86-848f-4d7a-9b62-e64b79749134","resolution":{"observed_at":"2026-06-29T15:03:32.176011Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":"2305.07895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-07-01T22:26:17.944984Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","venue":"cs.CV","work_id":"521163e9-4c77-4c73-8b7f-9a6aa75b6d3b","year":2023},"citing_paper":{"arxiv_id":"2606.29805","last_updated":"2026-06-30T14:53:46Z","snapshot_observed_at":"2026-07-07T00:03:45.783349Z","submitted_at":"2026-06-29T05:33:22Z","title":"Clearer Sight, Fewer Lies: Oriented Pickup Preference Optimization for Multimodal Hallucination Mitigation","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-06-30T06:13:11.013395Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2606.29805"},"observation_digest":"sha256:139bf6524b6028bcd6f57d65134338c14d01965a9733328a329ee042a4d5b484","observation_id":"47c1cc66-1afe-447b-a24b-08ea20db74ef","resolution":{"observed_at":"2026-06-30T06:14:18.757035Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":"2305.07895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-07-01T22:26:17.944984Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","venue":"cs.CV","work_id":"521163e9-4c77-4c73-8b7f-9a6aa75b6d3b","year":2023},"citing_paper":{"arxiv_id":"2606.29805","last_updated":"2026-06-30T14:53:46Z","snapshot_observed_at":"2026-07-07T00:03:45.783349Z","submitted_at":"2026-06-29T05:33:22Z","title":"Clearer Sight, Fewer Lies: Oriented Pickup Preference Optimization for Multimodal Hallucination Mitigation","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-07-01T07:01:26.911110Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2606.29805"},"observation_digest":"sha256:cb329ffb31aef969ddf37edc59eebce6486afe7db2235baece74c17d20d34d87","observation_id":"848b7e5e-cab8-4c4c-8505-caa9ffa96c71","resolution":{"observed_at":"2026-07-01T07:05:28.659900Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-08-02T11:58:26.902599Z","title":"OCRBench : On the Hidden Mystery of OCR in Large Multimodal Models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.22586","last_updated":"2026-06-09T04:56:59Z","snapshot_observed_at":"2026-08-06T10:04:16.098099Z","submitted_at":"2026-06-09T04:56:59Z","title":"MM-ShiftKV: Decode-Aware Prefill-Stage KV Selection for Multimodal Large Language Models","version":1},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-02T11:58:26.902599Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2607.22586"},"observation_digest":"sha256:023445ae58d5af20ca04664fb8827c8d03d1e1279354974abed6651d6dc59243","observation_id":"cfb5dd04-0ff9-4297-ad0c-19a70376f45d","resolution":{"observed_at":"2026-08-02T11:58:26.902599Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-08-05T04:16:08.308103Z","title":"arXiv preprint arXiv:2305.07895 , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04010","last_updated":"2026-08-04T17:59:58Z","snapshot_observed_at":"2026-08-07T00:27:25.189355Z","submitted_at":"2026-08-04T17:59:58Z","title":"ParVL: Parallel Scaling and Expandable Compute Allocation for Multimodal LLMs","version":1},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-08-05T04:16:08.308103Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2608.04010"},"observation_digest":"sha256:946a4ca342bd13b4a0ae0e4a587c8ee15fd766f5c3adbac3faa6d1948d5c4ccb","observation_id":"1f9fc299-3122-49a1-989c-4ec3476f6d40","resolution":{"observed_at":"2026-08-05T04:16:08.308103Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-08-06T23:51:35.327728Z","title":"2024 , volume =","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.04454","last_updated":"2026-08-05T05:09:20Z","snapshot_observed_at":"2026-08-07T00:38:30.489422Z","submitted_at":"2026-08-05T05:09:20Z","title":"Beyond Global Routing Aggregation: Phase-Aware Expert Merging for MoE Vision-Language Models","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-06T23:51:35.327728Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2608.04454"},"observation_digest":"sha256:4c27a2160c2cd8bd275fa7365b512d860b393bfefeeb5d610a55cc1908e32909","observation_id":"f4a88e1c-c85a-44a8-9e53-8cf0c6ab979f","resolution":{"observed_at":"2026-08-06T23:51:35.327728Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-08-06T11:55:27.987289Z","title":"On the hidden mystery of ocr in large multimodal models.arXiv preprint arXiv:2305.07895, 2023b","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.05000","last_updated":"2026-08-05T16:09:25Z","snapshot_observed_at":"2026-08-07T00:42:58.122448Z","submitted_at":"2026-08-05T16:09:25Z","title":"Towards Physics of Multimodal Pretraining: Knowledge Flow, Modality Synergy, Early Unification, and Recipes","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-06T11:55:27.987289Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2608.05000"},"observation_digest":"sha256:ff709110a30f994d132bc06f6926f82179411f7c226efe0de56d7533460cdd4d","observation_id":"d55ea9c9-4061-40c9-88f1-afb3999a8e46","resolution":{"observed_at":"2026-08-06T11:55:27.987289Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2305.07895/citation-record","integrity":"/paper/2305.07895/integrity","json":"/paper/2305.07895/citation-record.json","paper":"/paper/2305.07895"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"07eb0a06-5091-41c0-b751-d00d8cff832a","year":2023},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:9133615f1bf6fde16b080e084822737bdaf2bdfb91871b599c46edde136aa18b","observation_id":"e3018f5f-746f-4874-bcf0-d30380bd631f","resolution":{"observed_at":"2026-05-17T09:55:35.965876Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T07:26:54.787661Z","title":"Gpt-4 technical report","venue":null,"work_id":"388f534c-855a-4366-b933-f07bf3e2db5f","year":2023},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:728fc25eda81ef982e98c3ceb498ad684b586bda09c4dd169034368ee4ab99bd","observation_id":"07edbad5-452c-4ba6-a96a-82be0263c840","resolution":{"observed_at":"2026-05-17T09:55:35.973245Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":"2302.13971","doi":"10.48550/arxiv.2302.13971","metadata_source":"pith","pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LLaMA: Open and Efficient Foundation Language Models","venue":"cs.CL","work_id":"c018fc23-6f3f-4035-9d02-28a2173b2b9d","year":2023},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:e3eef5af8e3af0be8efb53ea44e4ab18e77e621bbc7b865396d9d553d7310750","observation_id":"e4bb3035-939f-4f16-9c56-d636cfac8420","resolution":{"observed_at":"2026-05-17T09:55:35.613195Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-01T11:08:05.851253+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T11:08:05.851253+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Hashimoto","venue":null,"work_id":"bf3517b5-0f2c-4f46-bff1-8d74220ebc3f","year":2023},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:88437c4c1284a93fbcb8146e5a1d2ffec8aa6dc28aeaa26d4c97f148d2203538","observation_id":"17a410ba-cf93-4ee0-803b-10375fe0b523","resolution":{"observed_at":"2026-05-17T09:55:35.976382Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Vicuna: An open-source chatbot impressing gpt-4 with 90%* chatgpt quality","venue":null,"work_id":"67dc94e1-9c8e-4287-ae6c-979bce9614cf","year":2023},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:042405ecc20e369f526b7e62cefcce37b60beab1eaf6c35dd89505fa4c5406de","observation_id":"82a4572f-85cd-4ddf-9727-4b0aa25b4d8e","resolution":{"observed_at":"2026-05-17T09:55:35.979929Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03277","last_updated":"2023-04-06T17:58:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-06T17:58:09Z","title":"Instruction Tuning with GPT-4","version":1},"cited_work":{"arxiv_id":"2304.03277","doi":"10.48550/arxiv.2304.03277","metadata_source":"pith","pith_arxiv_id":"2304.03277","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Instruction Tuning with GPT-4","venue":"cs.CL","work_id":"fd515477-f9f1-48aa-9feb-a3308e7656bb","year":2023},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"cited_paper":"/paper/2304.03277","citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:6c6f2ed6821199019c8f65f08f4a2bca4cd4a0979a20df9d39c3ceed96eed2cc","observation_id":"098beee5-b217-4297-9a9d-ebda6c80029c","resolution":{"observed_at":"2026-05-17T09:55:35.580214Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Vision-language pre-training: Basics, recent advances, and future trends","venue":null,"work_id":"802007bb-a337-4b31-a28f-0e49e31f8169","year":2022},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:ba89028dbccc8ae4eed45637bfdf064080e1c441e7075bf995ae9c8093e0b524","observation_id":"24fa484d-b9c1-473f-a942-34a5474d0d58","resolution":{"observed_at":"2026-05-17T09:55:35.983591Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2103.00020","last_updated":"2021-02-26T19:04:58Z","snapshot_observed_at":"2026-07-06T10:45:03.059688Z","submitted_at":"2021-02-26T19:04:58Z","title":"Learning Transferable Visual Models From Natural Language Supervision","version":1},"cited_work":{"arxiv_id":"2103.00020","doi":"10.1021/acs.jcim.0c00174","metadata_source":"pith","pith_arxiv_id":"2103.00020","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Learning Transferable Visual Models From Natural Language Supervision","venue":"cs.CV","work_id":"6de86bb5-27bd-4d5c-8b89-967ebfc52659","year":2021},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"cited_paper":"/paper/2103.00020","citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:fbdbe3f9252bc3c8cdbbf2b6d9f794f44b8a8ea30889672f4e6d3d676dadc9cc","observation_id":"3580bb70-9748-442a-b2cd-b5a1e6346a01","resolution":{"observed_at":"2026-05-17T09:55:35.527784Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2111.11432","last_updated":"2021-11-22T18:59:55Z","snapshot_observed_at":"2026-07-06T12:11:02.119174Z","submitted_at":"2021-11-22T18:59:55Z","title":"Florence: A New Foundation Model for Computer Vision","version":1},"cited_work":{"arxiv_id":"2111.11432","doi":null,"metadata_source":"pith","pith_arxiv_id":"2111.11432","snapshot_observed_at":"2026-07-04T20:00:08.247992Z","title":"Florence: A New Foundation Model for Computer Vision","venue":"cs.CV","work_id":"99823072-36a8-4b10-9ef5-a7f91da74650","year":2021},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"cited_paper":"/paper/2111.11432","citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:6d6759dd1c25b49ebbc69ff9553c2151d141bc0035ac021a83d78003b659d707","observation_id":"57929d05-82c6-4922-bb9c-552f9b78c597","resolution":{"observed_at":"2026-05-17T09:55:35.544307Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2102.05918","last_updated":"2021-06-11T07:51:39Z","snapshot_observed_at":"2026-07-06T10:40:28.145954Z","submitted_at":"2021-02-11T10:08:12Z","title":"Scaling Up Visual and Vision-Language Representation Learning With Noisy Text Supervision","version":2},"cited_work":{"arxiv_id":"2102.05918","doi":"10.48550/arxiv.2102.05918","metadata_source":"arxiv_reference","pith_arxiv_id":"2102.05918","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Le, Yunhsuan Sung, Zhen Li, and Tom Duerig","venue":"arXiv (Cornell University)","work_id":"d28390f3-8b21-4b2a-a523-473a16c2e43a","year":2021},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"cited_paper":"/paper/2102.05918","citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:f448f44f34a392fc3c52605e13a35399a09b2370ae0d4ff1df971e281d4f36f0","observation_id":"f433eb90-0de5-46e5-b917-1700b2cb27ab","resolution":{"observed_at":"2026-05-17T09:55:35.565340Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"ELEV ATER: A benchmark and toolkit for evaluating language-augmented visual models","venue":null,"work_id":"c9d2ea29-e35b-44a0-8a13-29e05077979a","year":2022},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:fbe8cfc7265c4ce7dac91b500257ce45e30cb6f708495a13a37e6b7e92e4f8f6","observation_id":"00491194-2d90-47ab-a3a8-6f44e0431290","resolution":{"observed_at":"2026-05-17T09:55:35.987260Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.03378","last_updated":"2023-03-06T18:58:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-03-06T18:58:06Z","title":"PaLM-E: An Embodied Multimodal Language Model","version":1},"cited_work":{"arxiv_id":"2303.03378","doi":"10.48550/arxiv.2303.03378","metadata_source":"pith","pith_arxiv_id":"2303.03378","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PaLM-E: An Embodied Multimodal Language Model","venue":"cs.LG","work_id":"5b99811a-1d93-47e2-9d59-f4045a0b74a2","year":2023},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"cited_paper":"/paper/2303.03378","citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:65ec4761a1865ab9c5f32598cf13bbcbbc8f11e609d9df1da526248df1762827","observation_id":"f43a8c3e-77df-480f-a432-95bbaf898ee0","resolution":{"observed_at":"2026-05-17T09:55:35.587553Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-09T19:19:06.044727+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-09T19:19:06.044727+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Flamingo: a visual language model for few-shot learning","venue":null,"work_id":"e350d96e-749e-4ec5-874e-b4e5f3e6d517","year":2022},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:e18d2422f6e0952c289b139d989ef6ac0a7cfc0b3d3ea8eb2c50ff00464c97d3","observation_id":"8664965c-e037-428c-a021-d4d3626bb449","resolution":{"observed_at":"2026-05-17T09:55:35.991138Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2205.14100","last_updated":"2022-12-15T19:21:35Z","snapshot_observed_at":"2026-08-04T09:31:34.784365Z","submitted_at":"2022-05-27T17:03:38Z","title":"GIT: A Generative Image-to-text Transformer for Vision and Language","version":5},"cited_work":{"arxiv_id":"2205.14100","doi":"10.48550/arxiv.2205.14100","metadata_source":"pith","pith_arxiv_id":"2205.14100","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GIT: A Generative Image-to-text Transformer for Vision and Language","venue":"cs.CV","work_id":"45b0563f-7563-4f00-8da5-8be70ff803fb","year":2022},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"cited_paper":"/paper/2205.14100","citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:3405a4625f24222f3e272e55838b632c162f4cd20920156f46c4110ac3b3421f","observation_id":"d4eaf250-8b13-44ac-bf4a-c707fc68693d","resolution":{"observed_at":"2026-05-17T09:55:35.619906Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-05-25T07:53:51.898843+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-25T07:53:51.898843+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Visual instruction tuning","venue":null,"work_id":"a2628f23-eb28-4e09-a098-bce23f22dee1","year":2024},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:0e100933b39a52654333a38ec626a99889e167aafcaf8cb08e151bb7be0544e3","observation_id":"9288a657-3986-4b89-b586-ae557e76ee80","resolution":{"observed_at":"2026-05-17T09:55:35.995154Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gemini: A family of highly capable multimodal models","venue":null,"work_id":"693aeffc-bb89-4778-88de-1477ae359cf1","year":2023},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:4edc462d2a32f540dca256ecb09ed95a6a8478fef91fee7a4e0ca3bc0f58c6c8","observation_id":"836926e4-a651-4b87-a162-330670ef6d85","resolution":{"observed_at":"2026-05-17T09:55:35.998881Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gpt-4v(ision) system card","venue":null,"work_id":"cd6900e8-55e2-455b-b3e9-3aa1081299e9","year":2023},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:71759f144732747308e9ff68cc82ab50a554e0d8bb44a033c68f6af962e87973","observation_id":"2e742101-1681-4989-a038-64592e4be446","resolution":{"observed_at":"2026-05-17T09:55:36.002567Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"BLIP-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","venue":null,"work_id":"907ff1f6-7579-4f12-940d-dd211f652490","year":2023},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:150e6efbb4f6e861612c40ac5d1765ea931a33caa1f8495bb13469e374ede40b","observation_id":"02d38667-983e-44c3-a647-aa3f2c9e0b18","resolution":{"observed_at":"2026-05-17T09:55:36.006286Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Openflamingo, March 2023","venue":null,"work_id":"fe04ce32-7667-49a6-912b-ca2a272d8271","year":2023},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:6bb5458dc56783e6cc1320b77c2182a741084ad3c0e88205f4181faa2391c507","observation_id":"bcdb8a84-bff6-4e75-8a92-b0c35b794236","resolution":{"observed_at":"2026-05-17T09:55:36.010196Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Improved baselines with visual instruction tuning","venue":null,"work_id":"fed9729f-b681-44b1-b296-acc34f481c44","year":2024},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:5e1677e0aff80f8010fe965400f4a4430bc215058ef481826d9946420a45c495","observation_id":"9249f411-1131-4bbc-9b12-772af225134b","resolution":{"observed_at":"2026-05-17T09:55:36.013765Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"MiniGPT-4: Enhancing vision- language understanding with advanced large language models","venue":null,"work_id":"1b75041e-763a-45dd-8592-9b6de8042bdc","year":2023},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:c82fbf12e0ccf3c832b04b8ea74e3005a452ef2be3c69c3d3adc39cc1ce77b91","observation_id":"8e1af592-8e9d-4e5f-abd5-2d90e2f0a537","resolution":{"observed_at":"2026-05-17T09:55:36.017326Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"mPLUG-Owl: Modularization empowers large language models with multimodality","venue":null,"work_id":"85a15e3c-0057-4c8e-b96e-f9fa8f124e9e","year":2023},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:875dfc29c3936cce5655b2724c06fe8096d44ee9cdf76203cbb4cc0bcdf97357","observation_id":"c84dbdaf-9f2f-4de4-832c-9407c5944f77","resolution":{"observed_at":"2026-05-17T09:55:36.020821Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"mplug-owl2: Revolutionizing multi-modal large language model with modality collaboration","venue":null,"work_id":"01c0da61-27b4-43e4-88e1-12fb263dd741","year":2023},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:7245401c9df9b6001b9db1090bec67176cb9cb8f5c2689c8fc5f8dfd8ec252b6","observation_id":"1ea6d9e9-01b1-41d9-a755-2e9ab8b7f566","resolution":{"observed_at":"2026-05-17T09:55:36.026095Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Llavar: Enhanced visual instruction tuning for text-rich image understanding","venue":null,"work_id":"8b6501f3-dab2-4fa1-b83f-9de4832d1cbf","year":2023},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:e8833f438f3c9c3a606c72b0b4234234928170f6251e1d6636902e0b415ecced","observation_id":"1e73d944-daed-44e4-9e68-dc9be9ee7fc2","resolution":{"observed_at":"2026-05-17T09:55:36.031135Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Bliva: A simple multimodal llm for better handling of text-rich visual questions","venue":null,"work_id":"60c33ab6-2f6a-447f-a84a-30ea8663fe0a","year":2023},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:b541064764ced0602ebc8350932cb1ac98b5dcccbfc966256c4ea710bf78729e","observation_id":"dba26dbd-532c-426c-bccf-b00d667269f1","resolution":{"observed_at":"2026-05-17T09:55:36.034630Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Minigpt-v2: large language model as a unified interface for vision-language multi-task learning","venue":null,"work_id":"d0a76fef-dba1-4e66-b563-5c0635bae7b5","year":2023},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:69568679be9bbe8cbbcad704c0fc861dc8b31b3fd03ca4cc8abff4fadc4fac24","observation_id":"e94d17b4-0109-4011-998d-f868cc09b42c","resolution":{"observed_at":"2026-05-17T09:55:35.657476Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Unidoc: A universal large multimodal model for simultaneous text detection, recognition, spotting and understanding","venue":null,"work_id":"be56930c-1d63-40a6-9d5e-0969d683fc1e","year":2023},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:a9b924d85f2444378880e450a7dc6406a38d77dd8542458d9afb3543495dbb58","observation_id":"d447f721-8dfc-4afe-a391-131311c6f4fa","resolution":{"observed_at":"2026-05-17T09:55:35.661608Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Docpedia: Unleashing the power of large multimodal model in the frequency domain for versatile document understanding","venue":null,"work_id":"035e54cd-7778-49a6-a00f-a742daec6968","year":2023},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:b31f996fcbf47e92cc049308f8a4cd5969fe5c3360a94022ac81eb876196ed01","observation_id":"5db0f4ac-9700-48fc-8f17-94df52d0f876","resolution":{"observed_at":"2026-05-17T09:55:35.664937Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Monkey: Image resolution and text label are important things for large multi-modal models","venue":null,"work_id":"f0293a05-b2d4-47dd-a0e1-75a35b4a6c06","year":2023},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:5596ff9887d2ea256d7325474b0d21a484e97006d994ac1ac389f1f7a6780564","observation_id":"da05f7db-e620-4062-be3c-4c5a37830ce7","resolution":{"observed_at":"2026-05-17T09:55:35.668375Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":"3fccb7bf-4ab6-4e08-ad33-706a9cc4742e","year":2024},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:aaaeb1ae32827afbd8f8bb1f73025e5a2ac87e47340d3484632808c08c6c6d6c","observation_id":"848f6c2d-6e4c-4187-a48b-84d1a67a4052","resolution":{"observed_at":"2026-05-17T09:55:35.671862Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Vlmevalkit: An open-source toolkit for evaluating large multi-modality models","venue":null,"work_id":"79a9874c-ab6c-4500-b672-c9d4d1985753","year":2024},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:f97c25007ef647a7304b25b700ecc51d4a97aac83074c45f14af1fe2bededb62","observation_id":"2001f8e1-f91e-4716-a3a0-407371e62897","resolution":{"observed_at":"2026-05-17T09:55:35.676031Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Lmms-eval: Accelerating the development of large multimoal models, March 2024","venue":null,"work_id":"cd30f1b9-7dde-4c87-860b-d1640b8f2812","year":2024},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:6e4ab9ac28b06cbc5ef6f98105c98a9796a68daa6add5a4da82149ea167b6974","observation_id":"563b9252-29a1-4aa8-819e-876f6a9b4124","resolution":{"observed_at":"2026-05-17T09:55:35.679774Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Mitigating hallucination in large multi-modal models via robust instruction tuning","venue":null,"work_id":"5dd47748-0b06-4702-aaca-a97f95b4414a","year":2023},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:17afcdfb1e6ec63980b347b77057b9607cc7b2a0fb7738f1bb64c0b461385004","observation_id":"2f851b79-4be4-4efb-a909-bbbed5ee89aa","resolution":{"observed_at":"2026-05-17T09:55:35.683238Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Mmbench: Is your multi-modal model an all-around player?","venue":null,"work_id":"e0bfac91-d04d-476c-a74d-ee445136bb89","year":2023},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:9fd587f19221371203709ad9e3fedcc7c14cf80f4a9c35d8f10c71fa20cb83e2","observation_id":"ed231db8-94fe-4003-9587-c8f2289d1a6a","resolution":{"observed_at":"2026-05-17T09:55:35.687434Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Mme: A comprehensive evaluation benchmark for multimodal large language models","venue":null,"work_id":"29b8caa6-69b1-4988-b0a5-4bd23ea25094","year":2023},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:77da79c5bf8f3bec84f57e8a43a9e54295cb1a9fab50d8dc63dc69b74a23501f","observation_id":"ac14142c-f662-460e-8ea0-843941b1f168","resolution":{"observed_at":"2026-05-17T09:55:35.692670Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"3cb28dbf-4946-4cbb-b58e-edca5fd3db00","year":2019},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:54867bc04797e1f00fa60fdc091e467aa62208fe4ceb32916a05b6f1dd3ef70c","observation_id":"3086762f-0c4b-4f40-80ea-35bf5c91d79e","resolution":{"observed_at":"2026-05-17T09:55:35.696614Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Top-down and bottom-up cues for scene text recognition","venue":null,"work_id":"c8a2e8ee-6ad7-488c-96f7-8b28b5be5bed","year":2012},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:170835862ded7debef8a6b78019e6ab816fe9bda3b00c55780bc77180ce68ed1","observation_id":"627a2cf1-3a0d-4cc4-9630-7bcdf8e0e5f3","resolution":{"observed_at":"2026-05-17T09:55:35.700933Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"End-to-end scene text recognition using tree-structured models","venue":null,"work_id":"3c47c9fc-682e-4a6b-8501-680b2d68681b","year":2014},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:b94a29e27fad62994c50c092239a26acc585c16319aa7b201743e9a44d8f6ab1","observation_id":"73639994-ce34-42e9-af6d-b308030a6502","resolution":{"observed_at":"2026-05-17T09:55:35.704596Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Iwamura, Lluís Gómez i Bigorda, Sergi Robles Mestre, Joan Mas Romeu, David Fernández Mota, Jon Almazán, and Lluís-Pere de las Heras","venue":null,"work_id":"51df0f81-d2de-4731-abe6-7fab2362dded","year":2013},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:5985fed02f2f368f19859c61352095e5e9bdf405dd099e33f1431131056fd3dc","observation_id":"5ec98a99-1e1c-4c95-b34c-c36dbb4aeac7","resolution":{"observed_at":"2026-05-17T09:55:35.708034Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Ghosh, Andrew D","venue":null,"work_id":"27af7a93-bb51-423d-8ac2-2b5cb3a06ae5","year":2015},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:b1391de0317b1ae3ea31dc1ea88c04b00bcbf7bbcaee683af66f40cfba2b5de4","observation_id":"2e52ce42-8e10-45bc-8a27-b58336c89864","resolution":{"observed_at":"2026-05-17T09:55:35.711237Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Recognizing text with perspective distortion in natural scenes","venue":null,"work_id":"5fc8bf22-6eaa-450a-8ec5-e00685726715","year":2013},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:b739d7215beb99d53c7e7e0dee01bf583617885cb1adcc69c56ac3918d852003","observation_id":"c89673a7-7f2b-4b1f-97d7-ab1f972f0ac4","resolution":{"observed_at":"2026-05-17T09:55:35.714497Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"A robust arbitrary text detection system for natural scene images","venue":null,"work_id":"aa45fbd5-2c28-4810-88ad-7d1c9407dca1","year":2014},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:9cbf8de6ae672854f3b14af408d04e651bfa537b9fe8644be11736d2e5aab30d","observation_id":"03648dec-bf78-48ac-9bbf-e773b4d59cc0","resolution":{"observed_at":"2026-05-17T09:55:35.717607Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1601.07140","last_updated":"2016-06-19T23:52:14Z","snapshot_observed_at":"2026-07-06T04:44:09.937735Z","submitted_at":"2016-01-26T19:30:34Z","title":"COCO-Text: Dataset and Benchmark for Text Detection and Recognition in Natural Images","version":2},"cited_work":{"arxiv_id":"1601.07140","doi":null,"metadata_source":"pith","pith_arxiv_id":"1601.07140","snapshot_observed_at":"2026-07-04T16:59:57.589641Z","title":"COCO-Text: Dataset and Benchmark for Text Detection and Recognition in Natural Images","venue":"cs.CV","work_id":"ea6ff5e9-689c-41f0-921f-ad32f5e27198","year":2016},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"cited_paper":"/paper/1601.07140","citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:ab7ae7a24e9f052801fc05a77d995ad214c7e5e1b3b44fa460b41fd5c05b0129","observation_id":"f2d68b78-65dd-460f-9954-608872ca8435","resolution":{"observed_at":"2026-05-17T09:55:35.647328Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Curved scene text detection via transverse and longitudinal sequence connection","venue":null,"work_id":"2ec57044-e8a1-4476-a457-946c04729777","year":2019},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:bcfbe5bb4ffa3390993395fbb4b95f58d41186a8f27bc90e6ac2cd34c34742f2","observation_id":"2fef0d94-7771-47cf-954c-b983dd98d12f","resolution":{"observed_at":"2026-05-17T09:55:35.720696Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Total-Text: A comprehensive dataset for scene text detection and recognition","venue":null,"work_id":"1d8da549-a882-4ebe-b96a-ce7deba7bee5","year":2017},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:826153a63bc1bc42e7a811a2b6abc5b805523536186d9948fe1ae755f133bd30","observation_id":"9556c3af-5f54-489a-8dd6-f4f62458cbdf","resolution":{"observed_at":"2026-05-17T09:55:35.723695Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"From two to one: A new scene text recognizer with visual language modeling network","venue":null,"work_id":"27adcc96-9ce7-497b-9341-f5bef9013a54","year":2021},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:5ba9bb74dbd91335a9ec628e75f7e7e9c3931721923fa1a64d869dd0440e5f59","observation_id":"ec8ea28c-9b57-42ec-a484-10bbc122e68f","resolution":{"observed_at":"2026-05-17T09:55:35.727093Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Toward understanding WordArt: Corner- guided transformer for scene text recognition","venue":null,"work_id":"734bfd50-ea7c-464d-9e94-fd90d93c81ec","year":2022},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:5cf5f70ac02f1285aaea8e8f5e255e011108bd456a1eefdf699aadc92a37ab80","observation_id":"0ae92482-c156-4376-a133-c8237e19e2fe","resolution":{"observed_at":"2026-05-17T09:55:35.730529Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The iam-database: an english sentence database for offline handwriting recognition","venue":null,"work_id":"35a63523-c337-466e-ab32-78ab6315413b","year":2002},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:92ae392c65a288b9ae7d71fde49cf4d0aaee6ccc6ba3267168a9baaf4f9dcec2","observation_id":"919ae2a6-c91e-4937-9a0e-3e76f694bd03","resolution":{"observed_at":"2026-05-17T09:55:35.733625Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Icdar 2019 robust reading challenge on reading chinese text on signboard","venue":null,"work_id":"4a0fe2f7-6b2e-45e4-b33c-a38905da7350","year":2019},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:2a99c82fc814c3ef3b600cbc5aafc632b7edbe5ee293d32a478a7a19396c4041","observation_id":"377bba11-67a3-46dd-a850-27db16d59520","resolution":{"observed_at":"2026-05-17T09:55:35.736742Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Saavedra, David Contreras, Juan Manuel Barrios, and Luiz S","venue":null,"work_id":"f335b18a-ac75-457a-a5ba-e377888cb69d","year":2014},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:24592130dd4e7ab4c2a087655825d247f75f0c5ff9314e9e460aad319c94d135","observation_id":"6f1ba111-c716-4f7a-81c8-4edad17891ee","resolution":{"observed_at":"2026-05-17T09:55:35.739487Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Jawahar, Ernest Valveny, and Dimosthenis Karatzas","venue":null,"work_id":"e0003679-5b73-4bcf-9367-d9484c2318a6","year":2019},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:50188099962d358fc772c95eff3ec5195b7dfeea39d3b631ad0ec91a3c499019","observation_id":"2a407ae6-0d1f-4418-ae4a-9cf1436282b0","resolution":{"observed_at":"2026-05-17T09:55:35.742352Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Towards VQA models that can read","venue":null,"work_id":"b1192075-552e-4f47-ab20-58f622fdfd2a","year":2019},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:a5b69ae562b141a96ca355ae0c03703a0ef52438e2d3e609f1f5dc55cc2c4f91","observation_id":"013f38b8-93ec-4aca-afbb-e2d7e74be12b","resolution":{"observed_at":"2026-05-17T09:55:35.745426Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"OCR-VQA: visual question answering by reading text in images","venue":null,"work_id":"92d59537-4cdf-4dc0-8205-8586c93fb5c1","year":2019},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:1a3e3544803c00cd974fb9d6440b77721539017ce046331c8ef5d6914306d9ff","observation_id":"9cdccc9b-283f-435a-8335-edad530ddba7","resolution":{"observed_at":"2026-05-17T09:55:35.748495Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"On the general value of evidence, and bilingual scene-text visual question answering","venue":null,"work_id":"45b828db-2a8c-4427-87a1-a8cb1a70f95f","year":2020},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:8e0db3048947ee523c30bae778a918d844083d4982dcd9ab549f3f99ae963e4f","observation_id":"43cfddb5-0de4-41a5-a6bf-5171561f672f","resolution":{"observed_at":"2026-05-17T09:55:35.751512Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"19f7242f-57ed-4402-9043-2bb10212dc1d","year":2021},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:55ebdc2cffb84ad5c0de55ef7c55a364ec82a3843a38d9f7d7c914788b054d3b","observation_id":"d9debcff-bccb-4618-b042-5f904bcf73a8","resolution":{"observed_at":"2026-05-17T09:55:35.754346Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Info- graphicvqa","venue":null,"work_id":"7c0881ce-be01-475a-9171-af9262381289","year":2022},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:12da46d594b534c88f1ef989150dec78bac84f6f53b1d9f183f4f589bbbce456","observation_id":"5a34f3cd-55f2-40eb-a328-fdb7f9093559","resolution":{"observed_at":"2026-05-17T09:55:35.757062Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.10244","last_updated":"2022-03-19T05:00:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-03-19T05:00:30Z","title":"ChartQA: A Benchmark for Question Answering about Charts with Visual and Logical Reasoning","version":1},"cited_work":{"arxiv_id":"2203.10244","doi":null,"metadata_source":"pith","pith_arxiv_id":"2203.10244","snapshot_observed_at":"2026-07-04T16:39:57.602645Z","title":"ChartQA: A Benchmark for Question Answering about Charts with Visual and Logical Reasoning","venue":"cs.CL","work_id":"8b49b78c-7e1d-4f57-af10-7c11cd63ff7c","year":2022},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"cited_paper":"/paper/2203.10244","citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:49c5d1ee211f3cb81e17288e225c572591df53b467763dbc82291be4da53099b","observation_id":"72e9551f-9aef-476f-921a-d5b9f8307f95","resolution":{"observed_at":"2026-05-17T09:55:35.606962Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"582db9ae-e3d8-42ff-aaa8-5e6cbbc1881a","year":2019},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:55970750f3925255525e80c02239b75d93c6f168c806028d33dc789ae2847401","observation_id":"4447d994-a44f-4499-9170-ee24e4e02ea7","resolution":{"observed_at":"2026-05-17T09:55:35.759809Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"FUNSD: A dataset for form understanding in noisy scanned documents","venue":null,"work_id":"20afa39f-f1f6-4704-a11a-218954277d60","year":2019},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:5b1f80583e8decbe292f1cc373557ebd3c6a0d1540bc5517d656646f8296e79c","observation_id":"7f809125-0343-4e50-94d3-06016436e286","resolution":{"observed_at":"2026-05-17T09:55:35.762948Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07498","last_updated":"2023-06-15T03:31:12Z","snapshot_observed_at":"2026-07-06T15:26:31.066492Z","submitted_at":"2023-05-12T14:11:47Z","title":"Visual Information Extraction in the Wild: Practical Dataset and End-to-end Solution","version":2},"cited_work":{"arxiv_id":"2305.07498","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.07498","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Visual information extraction in the wild: Practical dataset and end-to-end solution","venue":null,"work_id":"bb3440f4-176b-4ac6-a830-d6135e525af6","year":2023},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"cited_paper":"/paper/2305.07498","citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:2a05d569ad99770e62a6b22f285c6a66e58bd8bca9ff5b2f1a3d2d8240e4b44f","observation_id":"1c560636-1eb4-4213-a991-b454bc66eda0","resolution":{"observed_at":"2026-05-17T09:55:35.625941Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Syntax-aware network for handwritten mathematical expression recognition","venue":null,"work_id":"17f026df-6014-4457-b09a-2c0b40186bbd","year":2022},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:6e6915ae113df17210a138883f4fead9788d481c167b4f9a315dd3009159c0b4","observation_id":"2962206f-f949-43dd-80a7-73e09c356f49","resolution":{"observed_at":"2026-05-17T09:55:35.766244Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Minicpm-v2.6","venue":null,"work_id":"cbdbec94-d81f-49fe-9c50-6b529fe411a3","year":2024},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:2b8af32ffa796aef252edd9ac9ec71bfd4a0054940011bf8287b685e6cb649b4","observation_id":"206f8ebc-e791-420a-b980-0d4825360375","resolution":{"observed_at":"2026-05-17T09:55:35.769046Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Cambrian-1: A fully open, vision-centric exploration of multimodal llms","venue":null,"work_id":"7df457fc-b1ba-4a97-9fe2-a77879c47a4b","year":2024},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:9b0733a1fc8fc04ba85ed36bab3776addd33d6c276dc01c4cb65921c21209f0f","observation_id":"ad68df80-045d-422a-915e-95680111d5b8","resolution":{"observed_at":"2026-05-17T09:55:35.772435Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"How far are we to gpt-4v? closing the gap to commercial multimodal models with open-source suites","venue":null,"work_id":"ea2a12c1-d2da-4ea9-81ab-bcfdd7f4f443","year":2024},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:2beb33ac81972d45dc508933c2391f24155423a733803d055e6dffe2f5b535b6","observation_id":"39f370c1-0c5b-494d-80e6-1a367b98b058","resolution":{"observed_at":"2026-05-17T09:55:35.775391Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Paligemma: A versatile 3b vlm for transfer","venue":null,"work_id":"9d7438fe-8d33-4d9f-8f22-c97d40149bca","year":2024},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:048edfa2958213b980d85cec76423823587722af787e6ee9c6debd00ce917cbc","observation_id":"c01dfa1d-d8c5-4950-ac82-ba4856a39513","resolution":{"observed_at":"2026-05-17T09:55:35.778694Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Congrong","venue":null,"work_id":"7849b1b7-2483-4124-a5b8-1e29bcc3b607","year":2024},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:dc0c468e44f38855cd7a8e8802a8c4d980cf8b54a5471bb80119d13ed6bee6f2","observation_id":"49b15f92-d2a1-482b-b60c-ff9be11eb5cc","resolution":{"observed_at":"2026-05-17T09:55:35.781786Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Cogvlm: Visual expert for pretrained language models","venue":null,"work_id":"b0ad998b-c971-40d6-b2da-c8bce3a9b3f9","year":2023},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:71be53366499f27d68646087b3c14a45f3f2a04baf3b8dfd105a549cc7ab0d0a","observation_id":"6809c099-8ba9-4063-9706-3da71df3128a","resolution":{"observed_at":"2026-05-17T09:55:35.785110Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Minicpm-v-2","venue":null,"work_id":"3b561f6f-956b-4d5a-a351-eff5d712b27d","year":2024},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:02adc88bcf2f39a4d864dd739d438290eb1362812a44891a338daa09ba034534","observation_id":"7c1ac38f-c895-49e1-ac6c-f0aa960406cc","resolution":{"observed_at":"2026-05-17T09:55:35.789511Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Mini-monkey: Alleviate the sawtooth effect by multi-scale adaptive cropping","venue":null,"work_id":"d55742ad-c1c8-49b6-8ca8-0025b4660c56","year":2024},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:70ad030df258d7a98eed5be70e47bc8791a211d6c4b5f2402999399cdd843fe7","observation_id":"6422a54c-1f23-4479-af88-3db4824401b4","resolution":{"observed_at":"2026-05-17T09:55:35.793526Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Claude3.5-sonnet","venue":null,"work_id":"24d0dbe8-d7db-4799-aafe-e068490b3dc7","year":2024},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:1f71d678f23a34e4f6643b0216afa4023b5ba8b179cc32878c557e468d38d918","observation_id":"9ac367d5-0862-4049-9ae1-08a013de6070","resolution":{"observed_at":"2026-05-17T09:55:35.796304Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T16:42:39.744861Z","title":"Llava-next: Improved reasoning, ocr, and world knowledge, January 2024","venue":null,"work_id":"a84f281e-7f3f-4b95-9aaf-e3ba559cbf78","year":2024},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:bdd69bb6d3584388f887406a2af7f2f8f5bc7cee8a746ddbbfca26f9374bce0d","observation_id":"bf636cd1-f2a1-4da2-a160-892edcd71225","resolution":{"observed_at":"2026-05-17T09:55:35.799708Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gpt-4o-mini-20240718","venue":null,"work_id":"4cadbe50-8b41-4858-a57c-2b99220477e0","year":2024},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:149bd5a972353ec427bcece1a9749a6d021d2c3e367853de2e48a53e4deb9fd1","observation_id":"5499c977-00b1-4d40-9167-f67006d50d70","resolution":{"observed_at":"2026-05-17T09:55:35.802411Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Internlm-xcomposer2: Mastering free-form text-image composition and comprehension in vision-language large model","venue":null,"work_id":"2a6b4826-4777-458a-8b31-9d7980b09950","year":2024},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:e93029a4512f62db2f7b2f3e15266d59961256858463e430e5a4e44304b76b27","observation_id":"7eb9b621-d7b9-4aa8-a6a4-397f82ab123d","resolution":{"observed_at":"2026-05-17T09:55:35.805512Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Rekaflash","venue":null,"work_id":"17d91269-2bc8-4d73-be01-2d95cf4da4e5","year":2024},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:4920110f74db5ff466ed3dc422ce4b52ca1c756f2e42b8faed777d80b4ae18c8","observation_id":"be0d6b20-14ed-4d63-a672-1fb8e121c972","resolution":{"observed_at":"2026-05-17T09:55:35.808246Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gemini models","venue":null,"work_id":"233b57cb-6415-4fa9-9b3a-15d8d49ddb2e","year":2024},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:fe8bb9b995cb02a151159eae78842ddf39b22c8b4366fcadfd4cb4c2db0e92ea","observation_id":"0ef3d80a-43bf-4689-b98c-f53b343f80c1","resolution":{"observed_at":"2026-05-17T09:55:35.810826Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Xverse-v","venue":null,"work_id":"3e80a3ca-e57e-4309-8a0a-907e094dc0c7","year":2024},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:1a50579f0e678706a5715f0353cc88408c0495db33daf26a217f0e89e5d58948","observation_id":"67258c97-de25-4f0a-8c8b-a10b661b43cb","resolution":{"observed_at":"2026-05-17T09:55:35.813634Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Ovis: Structural embedding alignment for multimodal large language model","venue":null,"work_id":"3d5abf37-0634-475f-a08e-039187c2ac21","year":2024},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:82e6460a30ce58f2ec8690a8104e5ac74d28468474687a05b80909e0525370eb","observation_id":"0b4827e9-74fb-4a3d-8506-cc9e8d2c43ee","resolution":{"observed_at":"2026-05-17T09:55:35.816521Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-07T11:03:40.474471Z","title":"Qwen-vl: A versatile vision-language model for understanding, localization, text reading, and beyond","venue":null,"work_id":"b5c6eee0-36d1-4d4a-8258-f4ef8294b341","year":2023},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:4958ce5db215a7a744ffb4bb3519a1fa5948c5fec5236107a52a0e8eb96f3ae9","observation_id":"37c1de9d-2f69-47d2-be34-a2315a8f64b7","resolution":{"observed_at":"2026-05-17T09:55:35.819997Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Minicpm-llama3-v-2.5","venue":null,"work_id":"3cdfe494-fa6e-40c5-a58d-9aa63ffdff01","year":2024},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:838a7fcf99bdcc9011878a4c181d158b99be9f50a645f0c43e2c4111b9859430","observation_id":"0d185779-2b47-4e86-bdf1-63872e22470a","resolution":{"observed_at":"2026-05-17T09:55:35.825349Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Generative multimodal models are in-context learners","venue":null,"work_id":"69408b9c-98f0-4e9b-8b1a-cddaff4d699f","year":2024},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:b276cc2de5803f3186c498618d493bc239e8f0025407d68e94d1d60d78bed4aa","observation_id":"9778d63b-71cc-4f18-a2e7-4faea2aa247f","resolution":{"observed_at":"2026-05-17T09:55:35.832226Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Deepseek-vl: Towards real-world vision-language understanding","venue":null,"work_id":"a6174319-c99a-4503-9125-8046f96e1955","year":2024},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:4c0e02babe1fa178fd1aa40ea20cb0c1e1a42dcd7d234b54351fd6f6df7de4f7","observation_id":"aca2d950-8efe-490a-91d4-da7d2f4aed16","resolution":{"observed_at":"2026-05-17T09:55:35.838280Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"330fb66f-4c97-473f-b6ab-5d80a4bec317","year":2024},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:82be547c202bd7412992e7ee535d50ab3a7faef6838e6e3018fa8c6b6aa6f03f","observation_id":"e558722c-224a-4080-9f0d-cf3396cb0708","resolution":{"observed_at":"2026-05-17T09:55:35.843728Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Omnilmm-12b","venue":null,"work_id":"53a56fd0-7e05-4efe-9cc4-9e3524a91b8c","year":2024},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:ab779e3e038d387d22cb24696fdaf56dbc3806ec5a666402aa683a8569838f65","observation_id":"914be1ea-e8b1-4b36-be50-e6b5929f9619","resolution":{"observed_at":"2026-05-17T09:55:35.846909Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Transcore-m","venue":null,"work_id":"acc677b5-f848-4280-9830-00a672a5ec7c","year":2024},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:01eabd81fea279335ef28cca973b8bc7eeb309aa6a387f3e922db27267572b7d","observation_id":"2e0422b9-3901-4d51-81b6-eb22c305184e","resolution":{"observed_at":"2026-05-17T09:55:35.849863Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Internlm-xcomposer-2.5: A versatile large vision language model supporting long-contextual input and output","venue":null,"work_id":"08cf3293-a908-4fa1-ab3d-634f83cd32f7","year":2024},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:fb9ed87eeacc69508ed298fc35d28abd1d95689941f19f7bb558f1b644e691fe","observation_id":"1c24fcbf-1347-44c9-8e42-176ceb08e495","resolution":{"observed_at":"2026-05-17T09:55:35.853668Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Xtuner: A toolkit for efficiently fine-tuning llm","venue":null,"work_id":"e2620909-f004-444e-a36a-1c6b816ddb3c","year":2023},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:719f6ec27a25ce5cb4c489406c10137a03bdf98d570b41fa7d8ab9bbed1bb519","observation_id":"b07d8331-1a41-441d-9c1a-fde5fbe2cf73","resolution":{"observed_at":"2026-05-17T09:55:35.857477Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Sharegpt4v: Improving large multi-modal models with better captions","venue":null,"work_id":"bae8a30a-d5b9-4e91-8f23-48a846a51506","year":2023},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:8eaba3ab23072ed5f35b65947028f0274d303f9a5d35579aef99002202e51c07","observation_id":"705fa39a-d9c8-4fa2-87e8-c6b44288fe28","resolution":{"observed_at":"2026-05-17T09:55:35.861278Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"ed05ca0e-852e-437d-847c-4d679cf9810d","year":2024},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:6e768d5b853f5de3961752e2699a19ce4baaeb53e04ac5e31553f0a203c91460","observation_id":"036368dd-ba99-46be-b835-797ca156a7bd","resolution":{"observed_at":"2026-05-17T09:55:35.864922Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Internlm- xcomposer2-4khd: A pioneering large vision-language model handling resolutions from 336 pixels to 4k hd","venue":null,"work_id":"773412a9-3bbe-4fb9-9d0a-9968bfca7e94","year":2024},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:304a083a23eacf21ef36ec68516c58a9e067f6eff776763b5cac1d79e753dc5d","observation_id":"07893c62-b878-4faf-8bcf-08cc1ad25942","resolution":{"observed_at":"2026-05-17T09:55:35.868730Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Minicpm-v","venue":null,"work_id":"38d84a73-7bea-4342-b526-73f5d8816884","year":2024},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:cc37fe07b95f16530a73b2a299e4fb6745b2f56deb8523be8536f1933ee1f13f","observation_id":"e89b1c26-b857-49c4-ae81-d436a19ef5b8","resolution":{"observed_at":"2026-05-17T09:55:35.871922Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Yi-vl-34b","venue":null,"work_id":"d9d3ff88-ca82-4e4b-b44e-9138e6b7086c","year":2024},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:0e5177cfdf132b090a26278d108754ad3b3e33ccab4c3f656528c03ce61e30d1","observation_id":"f15f50a0-0cef-4d70-92f7-8694eb3a5653","resolution":{"observed_at":"2026-05-17T09:55:35.875328Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Rush, Douwe Kiela, Matthieu Cord, and Victor Sanh","venue":null,"work_id":"178654a8-051e-4f64-84cd-66f1e1ec02b7","year":2023},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:f4f494d9ff6feef7e2d56c7510c3df3b06412eca48f2b830ca7bd829b650a3ae","observation_id":"605797dc-30a0-4450-a600-1aa5426ec823","resolution":{"observed_at":"2026-05-17T09:55:35.879071Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"2308a86f-dacf-456d-883d-4078d2f016ca","year":2024},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:20aa9f8964892bc54992e8e05f86daf50dd063e7e4e0426162bef44bba66c96a","observation_id":"0de47630-c282-4fb5-b384-bbf177434c65","resolution":{"observed_at":"2026-05-17T09:55:35.882007Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Phi-3-vision","venue":null,"work_id":"1770056b-c92a-4751-9504-99fc3f847418","year":2024},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:44c413fc4ee7837e7c703c69dd119a9c41046e63ccc03511498a025a1884c316","observation_id":"3114ac99-a731-46ed-8021-e36d04d8d11a","resolution":{"observed_at":"2026-05-17T09:55:35.884897Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Glm: General language model pretraining with autoregressive blank infilling","venue":null,"work_id":"40a2ed16-b8cf-4e79-894b-784b8ab83010","year":2022},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:1b3c7d7a4896b059e94c35ce8798f4967ee3e117570e9d58c222fc59ba9daf2c","observation_id":"90dc8ffc-3a77-47f6-a2cb-5766a4d912a6","resolution":{"observed_at":"2026-05-17T09:55:35.888424Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"d7c9a895-667a-49e6-9dbd-4015e5a5d1ba","year":2024},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":96,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:71b200019eba5e20a5ccd103f825e8f0baf9025c7505932f9176c72153f3bdc9","observation_id":"9733e218-e1ad-4483-9b27-bf453995dad0","resolution":{"observed_at":"2026-05-17T09:55:35.891969Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.01390","last_updated":"2023-08-07T17:53:09Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-02T19:10:23Z","title":"OpenFlamingo: An Open-Source Framework for Training Large Autoregressive Vision-Language Models","version":2},"cited_work":{"arxiv_id":"2308.01390","doi":"10.48550/arxiv.2308.01390","metadata_source":"pith","pith_arxiv_id":"2308.01390","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OpenFlamingo: An Open-Source Framework for Training Large Autoregressive Vision-Language Models","venue":"cs.CV","work_id":"87bfa84a-e663-4165-806f-93ef439d88d0","year":2023},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"cited_paper":"/paper/2308.01390","citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:cf145150100dbe3a18268f7740871751cc390cf9b304c23aa9e53ff33561c4fe","observation_id":"10745213-2515-4aa3-8bb8-aa8e46197cf2","resolution":{"observed_at":"2026-05-17T09:55:35.601331Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"What matters when building vision-language models?","venue":null,"work_id":"bfe05629-c817-4ca9-aa3c-11ae4d973799","year":2024},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":98,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:12c8360f297c3ea27668f6a1697f942302b6213f849b73208652057ba738875e","observation_id":"2d1afb06-53ce-4253-b6cb-4b0430dc808a","resolution":{"observed_at":"2026-05-17T09:55:35.896853Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Pandagpt: One model to instruction-follow them all","venue":null,"work_id":"a2746c4c-a3df-46f0-a216-d3f3c280a8c1","year":2023},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":99,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:664027e2aedb96a956d38bf7a53a1073c3092aadf7154d13ae33637de7916a47","observation_id":"f433bf8e-bef2-4af3-9b02-9b7c184d2ac2","resolution":{"observed_at":"2026-05-17T09:55:35.901356Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"4a8bd93d-2b4a-4356-be7b-3eeeaadff7b4","year":2024},"citing_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"reference_index":100,"source":"pdf_text","source_observed_at":"2026-05-17T09:55:35.452649Z"},"links":{"citing_paper":"/paper/2305.07895"},"observation_digest":"sha256:dca1cffa6efbd3360f74431be5ed4f7fe6605d67ba1556c97383044e72e6b33a","observation_id":"dfab4ecd-c443-4da5-ab19-2ddc3cf455b7","resolution":{"observed_at":"2026-05-17T09:55:35.906886Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","latest_version":7,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T15:26:46.489500Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":9,"verified_exact":10,"verified_fuzzy":80},"total_outbound_references":122},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 100 of 122 outbound references and 45 inbound Pith citation observations for arXiv:2305.07895."}