{"as_of":"2026-08-24T02:04:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:685b0c17427beea22690ee4a1548954b5edfaa1cb544ba5cba60edd5e2385f3c","coverage":[{"denominator":55,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":55,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-17T20:50:57.814634Z","state":"measured"},{"denominator":130,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":130,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-23T06:30:58.430688+00:00","state":"measured"},{"denominator":75,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":75,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T22:53:52.033271Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-08T20:35:34.407064Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":"2409.01704","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-07-08T20:35:34.407064Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","venue":"cs.CV","work_id":"67b1592f-02b6-4056-b3fe-cc594e484192","year":2024},"citing_paper":{"arxiv_id":"2409.18839","last_updated":"2024-09-27T15:35:15Z","snapshot_observed_at":"2026-08-14T02:54:31.558222Z","submitted_at":"2024-09-27T15:35:15Z","title":"MinerU: An Open-Source Solution for Precise Document Content Extraction","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-16T04:00:25.624430Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2409.18839"},"observation_digest":"sha256:3b6f5e56b351d4851c2324754fc6653ee7eeeaa23daa22bb96180443a9667a72","observation_id":"9b61352c-f9e7-4ba8-93b1-0343acf84d18","resolution":{"observed_at":"2026-05-17T20:50:57.985910Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":"2409.01704","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-07-08T20:35:34.407064Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","venue":"cs.CV","work_id":"67b1592f-02b6-4056-b3fe-cc594e484192","year":2024},"citing_paper":{"arxiv_id":"2410.21169","last_updated":"2026-04-04T17:04:02Z","snapshot_observed_at":"2026-08-21T20:21:47.603765Z","submitted_at":"2024-10-28T16:11:35Z","title":"Document Parsing Unveiled: Techniques, Challenges, and Prospects for Structured Information Extraction","version":5},"reference_index":256,"source":"pdf_text","source_observed_at":"2026-05-23T19:15:21.695801Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2410.21169"},"observation_digest":"sha256:d0e298469adbf16260b399e7a8e467c5945ee287c25f92efe62c51c1c34f9c40","observation_id":"1e395297-78b8-4680-a6cc-3f4442b7996d","resolution":{"observed_at":"2026-05-23T19:15:47.044086Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-08-12T17:33:45.046203Z","title":"General OCR Theory: To- wards OCR-2.0 via a Unified End-to-End Model,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.17835","last_updated":"2024-11-19T12:09:12Z","snapshot_observed_at":"2026-08-23T23:44:44.543463Z","submitted_at":"2024-11-19T12:09:12Z","title":"Arabic-Nougat: Fine-Tuning Vision Transformers for Arabic OCR and Markdown Extraction","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-12T17:33:45.046203Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2411.17835"},"observation_digest":"sha256:3077ee6cae45ef2cbe8f7c9a56d6ebdbee6dcb5fceba02cce8375c0b5dc36e4f","observation_id":"8c19a771-7244-4420-8605-7120f2b21645","resolution":{"observed_at":"2026-08-12T17:33:45.046203Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-08-12T05:10:10.389994Z","title":"General ocr theory: Towards ocr-2.0 via a unified end-to-end model","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.00681","last_updated":"2024-12-01T05:44:01Z","snapshot_observed_at":"2026-08-21T17:41:14.518465Z","submitted_at":"2024-12-01T05:44:01Z","title":"MIMIC: Multimodal Islamophobic Meme Identification and Classification","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-12T05:10:10.389994Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2412.00681"},"observation_digest":"sha256:10a77f9c602459e81727a9e777fd13e6015d0543281ae7371dff495ad1776c62","observation_id":"ed5a5282-cece-4f82-b510-45b796e7b417","resolution":{"observed_at":"2026-08-12T05:10:10.389994Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-08-11T23:47:10.254073Z","title":"General ocr theory: Towards ocr-2.0 via a unified end-to-end model","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-19T21:55:09.283644Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.254073Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:0347aaae649be5383c35fb7f2072a31c9c915c76417ade8e71a5dd2c5b6bd5b7","observation_id":"f1ddf2ba-4fa9-4ea4-a4de-3b59de155505","resolution":{"observed_at":"2026-08-11T23:47:10.254073Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-08-11T23:21:42.107756Z","title":"General ocr theory: Towards ocr-2.0 via a unified end-to-end model","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02592","last_updated":"2025-08-30T07:10:08Z","snapshot_observed_at":"2026-08-16T23:34:46.066896Z","submitted_at":"2024-12-03T17:23:47Z","title":"OCR Hinders RAG: Evaluating the Cascading Impact of OCR on Retrieval-Augmented Generation","version":4},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-11T23:21:42.107756Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2412.02592"},"observation_digest":"sha256:08c59d0d239f2a1a66234b8d88166fc1e0e33a99e84370a29bd8157610bd773a","observation_id":"132e1a50-9027-4c35-8248-00fea3a2b30d","resolution":{"observed_at":"2026-08-11T23:21:42.107756Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-08-11T20:13:47.706187Z","title":"General ocr theory: Towards ocr-2.0 via a unified end-to-end model","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.05983","last_updated":"2025-07-26T14:45:01Z","snapshot_observed_at":"2026-08-14T09:40:33.773143Z","submitted_at":"2024-12-08T16:10:42Z","title":"Chimera: Improving Generalist Model with Domain-Specific Experts","version":3},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-11T20:13:47.706187Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2412.05983"},"observation_digest":"sha256:f3dc2f2807301b49a223190db7b9bae3a3ac5def8851ffbb426b1ef4c2812c88","observation_id":"a24772f8-bc34-485a-bf2c-578634d568be","resolution":{"observed_at":"2026-08-11T20:13:47.706187Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-08-11T18:41:26.773703Z","title":"General ocr theory: Towards ocr-2.0 via a unified end-to-end model","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.07626","last_updated":"2025-03-25T06:19:32Z","snapshot_observed_at":"2026-08-20T06:43:07.266054Z","submitted_at":"2024-12-10T16:05:56Z","title":"OmniDocBench: Benchmarking Diverse PDF Document Parsing with Comprehensive Annotations","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-11T18:41:26.773703Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2412.07626"},"observation_digest":"sha256:104d87363ec87d0ffcbdf3e68cdf6f5d27c14d74072f0c1e078f3307d26ca2ad","observation_id":"631c7b43-e1b5-4cbf-b71e-308a40f6fbdd","resolution":{"observed_at":"2026-08-11T18:41:26.773703Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-08-11T14:04:51.777751Z","title":"Preprint, arXiv:2409.01704","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.12505","last_updated":"2025-05-22T07:59:33Z","snapshot_observed_at":"2026-08-20T17:03:30.019807Z","submitted_at":"2024-12-17T03:20:00Z","title":"DocFusion: A Unified Framework for Document Parsing Tasks","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-11T14:04:51.777751Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2412.12505"},"observation_digest":"sha256:d4cd6b062b4224ce994550923e4d054df2df858863a8afd001f20fbfe0fe07b3","observation_id":"3a165f9f-8af0-404f-9027-4c9205e448a9","resolution":{"observed_at":"2026-08-11T14:04:51.777751Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-08-11T13:58:43.722047Z","title":"General ocr theory: Towards ocr-2.0 via a unified end-to-end model","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.12606","last_updated":"2024-12-17T07:06:10Z","snapshot_observed_at":"2026-08-19T12:28:31.813610Z","submitted_at":"2024-12-17T07:06:10Z","title":"Multi-Dimensional Insights: Benchmarking Real-World Personalization in Large Multimodal Models","version":1},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-08-11T13:58:43.722047Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2412.12606"},"observation_digest":"sha256:2140921e6b9b332bd57a49e7be33052299114a00061fbacd6c78e60b945ccc1c","observation_id":"c2718c4b-f368-43ac-ba93-365dc2bff340","resolution":{"observed_at":"2026-08-11T13:58:43.722047Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-08-11T11:55:10.583946Z","title":"General ocr theory: Towards ocr-2.0 via a unified end-to-end model","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.14835","last_updated":"2024-12-19T13:25:39Z","snapshot_observed_at":"2026-08-15T00:18:31.213910Z","submitted_at":"2024-12-19T13:25:39Z","title":"Progressive Multimodal Reasoning via Active Retrieval","version":1},"reference_index":112,"source":"pdf_text","source_observed_at":"2026-08-11T11:55:10.583946Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2412.14835"},"observation_digest":"sha256:eb2bf4632c85222a6d3b3bdd5a3c5ceba74124ef79b5169f7534e3b539c9ac28","observation_id":"68c8970f-eb14-4f7e-8ff1-b6d4041993d8","resolution":{"observed_at":"2026-08-11T11:55:10.583946Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-08-10T23:21:42.678604Z","title":"General ocr theory: Towards ocr-2.0 via a unified end-to-end model","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.20631","last_updated":"2025-01-26T23:16:36Z","snapshot_observed_at":"2026-08-19T06:28:57.942342Z","submitted_at":"2024-12-30T00:40:35Z","title":"Slow Perception: Let's Perceive Geometric Figures Step-by-step","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-10T23:21:42.678604Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2412.20631"},"observation_digest":"sha256:26412ff605e72b5c6c0078a53036f730459eabb3f7f6c970de1d225c0341384a","observation_id":"247ff7ab-c972-486c-86ed-def90eea7d84","resolution":{"observed_at":"2026-08-10T23:21:42.678604Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":"2409.01704","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-07-08T20:35:34.407064Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","venue":"cs.CV","work_id":"67b1592f-02b6-4056-b3fe-cc594e484192","year":2024},"citing_paper":{"arxiv_id":"2501.00321","last_updated":"2025-06-05T02:59:05Z","snapshot_observed_at":"2026-08-12T17:21:52.298102Z","submitted_at":"2024-12-31T07:32:35Z","title":"OCRBench v2: An Improved Benchmark for Evaluating Large Multimodal Models on Visual Text Localization and Reasoning","version":2},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-05-17T20:33:26.613927Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2501.00321"},"observation_digest":"sha256:580c4eb90c2f894f747392d6dd2b871e737b6167e48d70199dedb4cd2f0f1753","observation_id":"8b2f4c51-28a2-4949-a9d1-5d03b6f70994","resolution":{"observed_at":"2026-05-17T20:50:57.985910Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-08-08T23:13:14.003979Z","title":"General OCR theory: Towards OCR-2.0 via a unified end-to-end model","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.04223","last_updated":"2025-02-06T17:07:22Z","snapshot_observed_at":"2026-08-15T12:03:11.012078Z","submitted_at":"2025-02-06T17:07:22Z","title":"\\'Eclair -- Extracting Content and Layout with Integrated Reading Order for Documents","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-08T23:13:14.003979Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2502.04223"},"observation_digest":"sha256:d9347a49d6f04e718f4b790bcb8380220257907cf328c003f477b17429ebefb3","observation_id":"b43f16ab-e725-4eaf-bf37-13af65de50bd","resolution":{"observed_at":"2026-08-08T23:13:14.003979Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-08-09T06:01:13.403437Z","title":"General ocr theory: Towards ocr-2.0 via a unified end-to-end model","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.04371","last_updated":"2025-02-05T11:28:11Z","snapshot_observed_at":"2026-08-18T19:00:49.919850Z","submitted_at":"2025-02-05T11:28:11Z","title":"PerPO: Perceptual Preference Optimization via Discriminative Rewarding","version":1},"reference_index":68,"source":"arxiv_source","source_observed_at":"2026-08-09T06:01:13.403437Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2502.04371"},"observation_digest":"sha256:1c6fb0dd6306fda1f61418fcef0f3024cfd22ead4fe29e69b065ad0120687ffb","observation_id":"b5814798-dcd5-4b2d-81b7-208ac9a90280","resolution":{"observed_at":"2026-08-09T06:01:13.403437Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-08-07T22:57:16.881380Z","title":"General ocr theory: Towards ocr-2.0 via a unified end-to-end model,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.09020","last_updated":"2025-02-13T07:16:16Z","snapshot_observed_at":"2026-08-19T13:06:59.804647Z","submitted_at":"2025-02-13T07:16:16Z","title":"EventSTR: A Benchmark Dataset and Baselines for Event Stream based Scene Text Recognition","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T22:57:16.881380Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2502.09020"},"observation_digest":"sha256:852ce3cdaae68cf120c6b144c56d0bef4eb31828fab93f68348eacc03d5ac875","observation_id":"7711f7ff-4629-48e4-bd52-a996a0bba7eb","resolution":{"observed_at":"2026-08-07T22:57:16.881380Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":"2409.01704","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-07-08T20:35:34.407064Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","venue":"cs.CV","work_id":"67b1592f-02b6-4056-b3fe-cc594e484192","year":2024},"citing_paper":{"arxiv_id":"2502.16982","last_updated":"2025-02-24T09:12:29Z","snapshot_observed_at":"2026-08-16T22:48:15.725816Z","submitted_at":"2025-02-24T09:12:29Z","title":"Muon is Scalable for LLM Training","version":1},"reference_index":71,"source":"arxiv_source","source_observed_at":"2026-05-11T23:02:51.656353Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2502.16982"},"observation_digest":"sha256:e4ad8e8ab97501ecdc2ccfb862d6fd9a77dbd3478642cb164a5fa94f07055b54","observation_id":"0b73bebb-7e9d-4dae-9ec4-25d3d96d9260","resolution":{"observed_at":"2026-05-17T20:50:57.985910Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":"2409.01704","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-07-08T20:35:34.407064Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","venue":"cs.CV","work_id":"67b1592f-02b6-4056-b3fe-cc594e484192","year":2024},"citing_paper":{"arxiv_id":"2504.11101","last_updated":"2026-05-06T07:49:45Z","snapshot_observed_at":"2026-08-14T23:54:01.881029Z","submitted_at":"2025-04-15T11:51:18Z","title":"Consensus Entropy: Harnessing Multi-VLM Agreement for Self-Verifying and Self-Improving OCR","version":4},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-22T20:31:34.074705Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2504.11101"},"observation_digest":"sha256:b7575410cdb3cab30959ef1fff2eb56ae7100cf95d55f55d4d3bc3067f9bc415","observation_id":"a27c7054-d39e-40cf-bb19-cae973a8309c","resolution":{"observed_at":"2026-05-22T20:32:04.643167Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-08-15T22:53:52.033271Z","title":"General ocr theory: Towards ocr-2.0 via a unified end-to-end model","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.06038","last_updated":"2025-05-09T13:35:25Z","snapshot_observed_at":"2026-08-20T09:44:30.954537Z","submitted_at":"2025-05-09T13:35:25Z","title":"Document Image Rectification Bases on Self-Adaptive Multitask Fusion","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-15T22:53:52.033271Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2505.06038"},"observation_digest":"sha256:56183726c894c1fcbf8dae2f981259f5fffe830041c3e1a47b860d17a1f40ce6","observation_id":"33026d3b-32b8-43b2-b4ad-aba31cdb9822","resolution":{"observed_at":"2026-08-15T22:53:52.033271Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-08-15T20:31:36.623999Z","title":"General ocr theory: Towards ocr-2.0 via a unified end-to-end model","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.12766","last_updated":"2025-05-19T06:45:18Z","snapshot_observed_at":"2026-08-19T13:06:54.634082Z","submitted_at":"2025-05-19T06:45:18Z","title":"Reasoning-OCR: Can Large Multimodal Models Solve Complex Logical Reasoning Problems from OCR Cues?","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-15T20:31:36.623999Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2505.12766"},"observation_digest":"sha256:ddeb7e073ef4b279b7b80446e0298c575224f3f195cfc5248bc483831082c82e","observation_id":"6141f1d2-1314-4115-9085-0329bb18c23e","resolution":{"observed_at":"2026-08-15T20:31:36.623999Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-08-07T15:42:29.746865Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.14059","last_updated":"2025-05-20T08:03:59Z","snapshot_observed_at":"2026-08-19T17:28:16.492186Z","submitted_at":"2025-05-20T08:03:59Z","title":"Dolphin: Document Image Parsing via Heterogeneous Anchor Prompting","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-07T15:42:29.746865Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2505.14059"},"observation_digest":"sha256:c07c81e63b22e1b509d67feee3ce90e668601f35183e9cb95b30257b90d01b86","observation_id":"05f74a1d-8b18-407b-b1f3-b588864851e2","resolution":{"observed_at":"2026-08-07T15:42:29.746865Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-08-06T20:39:37.205231Z","title":"General ocr theory: Towards ocr-2.0 via a unified end-to-end model,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02200","last_updated":"2025-07-02T23:41:31Z","snapshot_observed_at":"2026-08-14T12:20:51.035067Z","submitted_at":"2025-07-02T23:41:31Z","title":"ESTR-CoT: Towards Explainable and Accurate Event Stream based Scene Text Recognition with Chain-of-Thought Reasoning","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T20:39:37.205231Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2507.02200"},"observation_digest":"sha256:d8bd7030cb1f7fe8add801a547ece594316c2d4cfea3f20afe8c293ad9ee2546","observation_id":"a060338a-bfa2-405d-91f5-108686d35f23","resolution":{"observed_at":"2026-08-06T20:39:37.205231Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":"2409.01704","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-07-08T20:35:34.407064Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","venue":"cs.CV","work_id":"67b1592f-02b6-4056-b3fe-cc594e484192","year":2024},"citing_paper":{"arxiv_id":"2507.08458","last_updated":"2026-04-14T14:30:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-11T10:02:08Z","title":"A document is worth a structured record: Principled inductive bias design for document recognition","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-19T04:57:35.441758Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2507.08458"},"observation_digest":"sha256:706f70a1ec9602e218822e077390b6eb54bbfa9c8a09010f63061dafd5bb8685","observation_id":"20ca24ff-9389-43fb-b12d-ccb9e7061bbc","resolution":{"observed_at":"2026-05-19T05:02:04.440884Z","resolver_source":"local_arxiv","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-08-15T18:19:33.643325Z","title":"arXiv preprint arXiv:2409.01704","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.18264","last_updated":"2025-08-25T05:11:11Z","snapshot_observed_at":"2026-08-18T13:43:30.704189Z","submitted_at":"2025-07-24T10:08:43Z","title":"Zero-shot OCR Accuracy of Low-Resourced Languages: A Comparative Analysis on Sinhala and Tamil","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-15T18:19:33.643325Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2507.18264"},"observation_digest":"sha256:61f3ec45f52631698f13aedc18f6104273f0f78cd8f59dea849f561a3640bc33","observation_id":"80793239-78d1-4cc0-950b-73f4e181560d","resolution":{"observed_at":"2026-08-15T18:19:33.643325Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-08-15T18:20:41.300411Z","title":"General ocr theory: Towards ocr-2.0 via a unified end-to-end model","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.18300","last_updated":"2025-07-24T11:05:24Z","snapshot_observed_at":"2026-08-17T22:35:22.675252Z","submitted_at":"2025-07-24T11:05:24Z","title":"LMM-Det: Make Large Multimodal Models Excel in Object Detection","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-15T18:20:41.300411Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2507.18300"},"observation_digest":"sha256:e48945f0f76f3a612b4a04680d67a2c83b2ecab2680a2fb2ced5c3c13659c9ff","observation_id":"e79f9c03-8f8e-4163-8f48-d84cf427e9cc","resolution":{"observed_at":"2026-08-15T18:20:41.300411Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-08-06T12:10:08.420714Z","title":"General ocr theory: Towards ocr-2.0 via a unified end-to-end model.arXiv preprint arXiv:2409.01704, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.22058","last_updated":"2025-07-29T17:59:04Z","snapshot_observed_at":"2026-08-15T19:53:33.201174Z","submitted_at":"2025-07-29T17:59:04Z","title":"X-Omni: Reinforcement Learning Makes Discrete Autoregressive Image Generative Models Great Again","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-06T12:10:08.420714Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2507.22058"},"observation_digest":"sha256:d0fd67ec751fc8ff7820292bac721ba7309ae6078fbd69f48e99ecd70db7d3d7","observation_id":"ae9a255d-3e03-438a-8fe1-ed02d81dea12","resolution":{"observed_at":"2026-08-06T12:10:08.420714Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-08-05T20:20:12.582337Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.10681","last_updated":"2025-08-14T14:24:47Z","snapshot_observed_at":"2026-08-06T15:57:51.802224Z","submitted_at":"2025-08-14T14:24:47Z","title":"IADGPT: Unified LVLM for Few-Shot Industrial Anomaly Detection, Localization, and Reasoning via In-Context Learning","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-05T20:20:12.582337Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2508.10681"},"observation_digest":"sha256:61d39177a8a64126967e088a474e46e10f10c2aa2cabcfb19a14822e78bd4d9a","observation_id":"e7445f10-8aee-4e75-9637-f528bf00acaa","resolution":{"observed_at":"2026-08-05T20:20:12.582337Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-08-05T10:52:13.591046Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.03615","last_updated":"2025-09-03T18:08:41Z","snapshot_observed_at":"2026-08-17T10:54:25.988526Z","submitted_at":"2025-09-03T18:08:41Z","title":"E-ARMOR: Edge case Assessment and Review of Multilingual Optical Character Recognition","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-05T10:52:13.591046Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2509.03615"},"observation_digest":"sha256:66ac90fbb35e40e43104346fd266dc068c0e957ddef0c6bf197b26f871375c17","observation_id":"f345069e-23ab-46ec-b1b7-be2235b54907","resolution":{"observed_at":"2026-08-05T10:52:13.591046Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-08-15T16:26:48.764056Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.07666","last_updated":"2025-09-06T00:59:28Z","snapshot_observed_at":"2026-08-18T20:09:11.531709Z","submitted_at":"2025-09-06T00:59:28Z","title":"MoLoRAG: Bootstrapping Document Understanding via Multi-modal Logic-aware Retrieval","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-15T16:26:48.764056Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2509.07666"},"observation_digest":"sha256:ff9040029be0553e09e46b0df7844c164a3f36d9b718f734d144cae2d26c37f4","observation_id":"71dc6786-089a-44d0-af56-97bc2990d45f","resolution":{"observed_at":"2026-08-15T16:26:48.764056Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":"2409.01704","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-07-08T20:35:34.407064Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","venue":"cs.CV","work_id":"67b1592f-02b6-4056-b3fe-cc594e484192","year":2024},"citing_paper":{"arxiv_id":"2509.22186","last_updated":"2025-09-29T16:41:28Z","snapshot_observed_at":"2026-08-15T07:34:12.195990Z","submitted_at":"2025-09-26T10:45:48Z","title":"MinerU2.5: A Decoupled Vision-Language Model for Efficient High-Resolution Document Parsing","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-17T13:25:31.884175Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2509.22186"},"observation_digest":"sha256:9bcc40efec004b0e6c68fb9e27237ac7fec127ec314fdebfac78dd6b8f9c7b8e","observation_id":"765ad383-e252-4db4-9a90-bec6d7788331","resolution":{"observed_at":"2026-05-17T20:50:57.985910Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":"2409.01704","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-07-08T20:35:34.407064Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","venue":"cs.CV","work_id":"67b1592f-02b6-4056-b3fe-cc594e484192","year":2024},"citing_paper":{"arxiv_id":"2510.18234","last_updated":"2025-10-21T02:41:44Z","snapshot_observed_at":"2026-08-17T02:37:33.939393Z","submitted_at":"2025-10-21T02:41:44Z","title":"DeepSeek-OCR: Contexts Optical Compression","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-12T04:35:50.647950Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2510.18234"},"observation_digest":"sha256:d7b6d9f0b6ac2f369bedfd9a7c8eb41a0be5bd651024db079ae849a1ae062078","observation_id":"756a901f-d8bc-42ea-b6de-7714c6a759cd","resolution":{"observed_at":"2026-05-17T20:50:57.985910Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":"2409.01704","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-07-08T20:35:34.407064Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","venue":"cs.CV","work_id":"67b1592f-02b6-4056-b3fe-cc594e484192","year":2024},"citing_paper":{"arxiv_id":"2511.02014","last_updated":"2026-05-21T09:12:23Z","snapshot_observed_at":"2026-08-18T05:26:36.340013Z","submitted_at":"2025-11-03T19:39:13Z","title":"Towards Selection of Large Multimodal Models as Engines for Burned-in Protected Health Information Detection in Medical Images","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-22T13:25:37.515812Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2511.02014"},"observation_digest":"sha256:53447cb0fcb7154ebfa095a1476e2a18cbc8d2e3d89dfc84041a1738bd854e56","observation_id":"18d8d4c7-1379-42a6-a82a-31ad7dca5d47","resolution":{"observed_at":"2026-05-22T13:26:35.645223Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":"2409.01704","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-07-08T20:35:34.407064Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","venue":"cs.CV","work_id":"67b1592f-02b6-4056-b3fe-cc594e484192","year":2024},"citing_paper":{"arxiv_id":"2511.14998","last_updated":"2026-04-07T03:13:19Z","snapshot_observed_at":"2026-08-15T01:12:11.347044Z","submitted_at":"2025-11-19T00:41:14Z","title":"FinCriticalED: A Visual Benchmark for Financial Fact-Level OCR","version":3},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-17T20:19:52.701262Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2511.14998"},"observation_digest":"sha256:6b44ebe35fd5079f09fa12db7c0b80892b95c4cbe4dca809f7d882247c6936fc","observation_id":"605d2982-70bb-4c1e-8f10-895ff9f74128","resolution":{"observed_at":"2026-05-17T20:50:57.985910Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-08-03T20:15:30.541979Z","title":"General ocr theory: Towards ocr-2.0 via a unified end-to-end model.arXiv preprint arXiv:2409.01704,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2511.20651","last_updated":"2026-07-21T00:23:00Z","snapshot_observed_at":"2026-08-14T13:42:29.127732Z","submitted_at":"2025-11-25T18:59:55Z","title":"RubricRL: Simple Generalizable Rewards for Text-to-Image Generation","version":4},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-03T20:15:30.541979Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2511.20651"},"observation_digest":"sha256:cc4e6d70290db26d1f4fe3df4d9e1b682c892acedad5817ac4e07dfa0c6faa5d","observation_id":"7f16cff5-3871-436d-93a9-2626f4638884","resolution":{"observed_at":"2026-08-03T20:15:30.541979Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-08-03T14:16:50.859133Z","title":"General ocr theory: Towards ocr-2.0 via a unified end-to-end model.arXiv preprint arXiv:2409.01704, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2512.21095","last_updated":"2026-07-11T09:04:48Z","snapshot_observed_at":"2026-08-19T23:40:02.808027Z","submitted_at":"2025-12-24T10:35:21Z","title":"UniRec-0.1B: Unified Text and Formula Recognition with 0.1B Parameters","version":2},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-03T14:16:50.859133Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2512.21095"},"observation_digest":"sha256:76389ac5827c64d8b4515e659237775cdba1810fe818314b2a97062069d60b18","observation_id":"b19ebe6a-47ea-4785-8136-98a70a01bf87","resolution":{"observed_at":"2026-08-03T14:16:50.859133Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":"2409.01704","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-07-08T20:35:34.407064Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","venue":"cs.CV","work_id":"67b1592f-02b6-4056-b3fe-cc594e484192","year":2024},"citing_paper":{"arxiv_id":"2601.04068","last_updated":"2026-05-20T06:08:56Z","snapshot_observed_at":"2026-08-16T10:53:06.656788Z","submitted_at":"2026-01-07T16:32:17Z","title":"Mind the Generative Details: Direct Localized Detail Preference Optimization for Video Diffusion Models","version":3},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-05-16T16:25:03.743594Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2601.04068"},"observation_digest":"sha256:2761998183979f20bb7851da1096af8e7624b2216596c109c34b60e1c34e3a20","observation_id":"410fd685-5f7b-400b-9dc9-880c8072c599","resolution":{"observed_at":"2026-05-17T20:50:57.985910Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":"2409.01704","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-07-08T20:35:34.407064Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","venue":"cs.CV","work_id":"67b1592f-02b6-4056-b3fe-cc594e484192","year":2024},"citing_paper":{"arxiv_id":"2601.04068","last_updated":"2026-05-20T06:08:56Z","snapshot_observed_at":"2026-08-16T10:53:06.656788Z","submitted_at":"2026-01-07T16:32:17Z","title":"Mind the Generative Details: Direct Localized Detail Preference Optimization for Video Diffusion Models","version":4},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-05-21T16:01:52.150950Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2601.04068"},"observation_digest":"sha256:8a6f1970cfcb9ae1dbca32165d1004d13765a1b51116a32a37af63b560b83b0f","observation_id":"d394d786-bcc7-45b0-ae78-a5bd01e36e7f","resolution":{"observed_at":"2026-05-21T16:04:14.700455Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":"2409.01704","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-07-08T20:35:34.407064Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","venue":"cs.CV","work_id":"67b1592f-02b6-4056-b3fe-cc594e484192","year":2024},"citing_paper":{"arxiv_id":"2601.09298","last_updated":"2026-05-07T06:40:25Z","snapshot_observed_at":"2026-08-07T01:21:01.981841Z","submitted_at":"2026-01-14T09:01:46Z","title":"Multi-Modal LLM based Image Captioning in ICT: Bridging the Gap Between General and Industry Domain","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-16T14:26:08.188950Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2601.09298"},"observation_digest":"sha256:b9f17713266a6bc663399580c05db46d384d53eeeb2a867e42adf390cde38b59","observation_id":"f6ed3d2e-45d9-4b4e-95a2-1727ff1ccdda","resolution":{"observed_at":"2026-05-17T20:50:57.985910Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-08-15T15:44:17.198748Z","title":"General ocr theory: Towards ocr-2.0 via a unified end-to-end model","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2601.20430","last_updated":"2026-07-13T14:16:50Z","snapshot_observed_at":"2026-08-21T20:22:10.786691Z","submitted_at":"2026-01-28T09:37:13Z","title":"Youtu-Parsing: Perception, Structuring and Recognition via High-Parallelism Decoding","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-15T15:44:17.198748Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2601.20430"},"observation_digest":"sha256:40bc318d317578c0cc5ff05c59da4f08b47cda0e438411024565674ceab0b5bf","observation_id":"8374201a-4916-4d89-8531-ceb6b2fd6130","resolution":{"observed_at":"2026-08-15T15:44:17.198748Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":"2409.01704","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-07-08T20:35:34.407064Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","venue":"cs.CV","work_id":"67b1592f-02b6-4056-b3fe-cc594e484192","year":2024},"citing_paper":{"arxiv_id":"2602.01785","last_updated":"2026-08-14T09:39:49Z","snapshot_observed_at":"2026-08-19T23:09:06.003911Z","submitted_at":"2026-02-02T08:10:21Z","title":"Seeing is Coding: On the Effectiveness of Vision Language Models in Code Understanding","version":2},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-05-16T08:30:50.984873Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2602.01785"},"observation_digest":"sha256:8d7e52a769696ce7f3c6ad85bdbcbcb1b2fe7a7861f2eb613197ed692d0dbd3c","observation_id":"2b592335-508d-43c3-af07-c65e28a0ddd3","resolution":{"observed_at":"2026-05-17T20:50:57.985910Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-08-02T23:44:40.099054Z","title":"arXiv preprint arXiv:2409.01704 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.12957","last_updated":"2026-06-29T06:43:50Z","snapshot_observed_at":"2026-08-14T14:25:16.701918Z","submitted_at":"2026-02-13T14:22:10Z","title":"HSD: Training-Free Acceleration for Document Parsing Vision-Language Models with Hierarchical Speculative Decoding","version":3},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-02T23:44:40.099054Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2602.12957"},"observation_digest":"sha256:b5fdb28c6f84fa0d4f56be6f3dc2641e45cda95d0d0800c8f1c1595290b57ad6","observation_id":"45bc09ba-417f-4053-9ca0-68d1e493c820","resolution":{"observed_at":"2026-08-02T23:44:40.099054Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-08-02T22:27:43.628538Z","title":"General ocr theory: Towards ocr-2.0 via a unified end-to-end model, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.16872","last_updated":"2026-05-27T14:51:57Z","snapshot_observed_at":"2026-08-19T19:03:00.291745Z","submitted_at":"2026-02-18T20:59:22Z","title":"DODO: Discrete OCR Diffusion Models","version":2},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-02T22:27:43.628538Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2602.16872"},"observation_digest":"sha256:fd1ce716eb181804d60ef44c0b3e91247f1b1b87d519920a41067ac27b0ac92e","observation_id":"a9f30a76-cbde-4426-8a68-f639683e31eb","resolution":{"observed_at":"2026-08-02T22:27:43.628538Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":"2409.01704","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-07-08T20:35:34.407064Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","venue":"cs.CV","work_id":"67b1592f-02b6-4056-b3fe-cc594e484192","year":2024},"citing_paper":{"arxiv_id":"2603.23885","last_updated":"2026-04-18T12:13:54Z","snapshot_observed_at":"2026-08-11T00:45:34.952402Z","submitted_at":"2026-03-25T03:19:09Z","title":"Towards Real-World Document Parsing via Realistic Scene Synthesis and Document-Aware Training","version":3},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-15T01:15:26.757215Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2603.23885"},"observation_digest":"sha256:4d36564e3791aac2b234ae80d59feca86bbbcffede93c2582b1fe757a7959cab","observation_id":"1bba8d35-e6f2-42cb-af78-bdc67548e64e","resolution":{"observed_at":"2026-05-17T20:50:57.985910Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":"2409.01704","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-07-08T20:35:34.407064Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","venue":"cs.CV","work_id":"67b1592f-02b6-4056-b3fe-cc594e484192","year":2024},"citing_paper":{"arxiv_id":"2603.24326","last_updated":"2026-04-03T06:49:01Z","snapshot_observed_at":"2026-07-06T22:50:28.640006Z","submitted_at":"2026-03-25T14:08:56Z","title":"Boosting Document Parsing Efficiency and Performance with Coarse-to-Fine Visual Processing","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-15T00:25:19.782732Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2603.24326"},"observation_digest":"sha256:02ea950d5eb487c9b46f983495e2592d8602756f2213c3e95309520641228b9d","observation_id":"7ec6c456-bd98-46dc-a400-ca6449cd05c3","resolution":{"observed_at":"2026-05-17T20:50:57.985910Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":"2409.01704","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-07-08T20:35:34.407064Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","venue":"cs.CV","work_id":"67b1592f-02b6-4056-b3fe-cc594e484192","year":2024},"citing_paper":{"arxiv_id":"2604.00270","last_updated":"2026-06-05T03:05:20Z","snapshot_observed_at":"2026-08-20T01:56:44.062129Z","submitted_at":"2026-03-31T21:51:36Z","title":"OmniSch: A Multimodal PCB Schematic Benchmark For Structured Diagram Visual Reasoning","version":3},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-13T23:17:45.005518Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2604.00270"},"observation_digest":"sha256:8d76cb9434717f9ef29e464fe05a5ab1bf93b150404a8840405fdf7b427958f4","observation_id":"9c01214c-a45a-4446-a616-1825defcb246","resolution":{"observed_at":"2026-05-17T20:50:57.985910Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-07-13T15:15:25.977993Z","title":"arXiv preprint arXiv:2409.01704 (2024), https://arxiv.org/abs/2409.01704","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2604.00270","last_updated":"2026-06-05T03:05:20Z","snapshot_observed_at":"2026-08-20T01:56:44.062129Z","submitted_at":"2026-03-31T21:51:36Z","title":"OmniSch: A Multimodal PCB Schematic Benchmark For Structured Diagram Visual Reasoning","version":4},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-07-13T15:15:25.977993Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2604.00270"},"observation_digest":"sha256:e9e98baf14659ff009d91cde3de208a8044b3e2e3de7844aeb28f66aae88619d","observation_id":"0cd11a8a-a797-40f9-8401-9826349b4258","resolution":{"observed_at":"2026-07-13T15:15:25.977993Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":"2409.01704","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-07-08T20:35:34.407064Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","venue":"cs.CV","work_id":"67b1592f-02b6-4056-b3fe-cc594e484192","year":2024},"citing_paper":{"arxiv_id":"2604.02880","last_updated":"2026-04-17T07:24:14Z","snapshot_observed_at":"2026-08-15T15:48:23.864869Z","submitted_at":"2026-04-03T08:44:45Z","title":"InstructTable: Improving Table Structure Recognition Through Instructions","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-13T20:53:57.029294Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2604.02880"},"observation_digest":"sha256:8e19f6e367350f8319487e80947ca6b0c813de26aaf01828af6def1ee4a082b5","observation_id":"69a4efdd-10d5-4355-9efb-0c25a196660a","resolution":{"observed_at":"2026-05-17T20:50:57.985910Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":"2409.01704","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-07-08T20:35:34.407064Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","venue":"cs.CV","work_id":"67b1592f-02b6-4056-b3fe-cc594e484192","year":2024},"citing_paper":{"arxiv_id":"2604.04771","last_updated":"2026-04-09T09:16:34Z","snapshot_observed_at":"2026-08-12T11:17:57.847430Z","submitted_at":"2026-04-06T15:44:18Z","title":"MinerU2.5-Pro: Pushing the Limits of Data-Centric Document Parsing at Scale","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-10T18:58:41.377996Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2604.04771"},"observation_digest":"sha256:6619fdef941401cd83e4eeaadcfabdd6e8c2e016274b88f8e4ffe8757fab8f7e","observation_id":"780c6f79-2f43-4090-be7a-0de66f9ae74d","resolution":{"observed_at":"2026-05-17T20:50:57.985910Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":"2409.01704","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-07-08T20:35:34.407064Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","venue":"cs.CV","work_id":"67b1592f-02b6-4056-b3fe-cc594e484192","year":2024},"citing_paper":{"arxiv_id":"2604.16070","last_updated":"2026-04-17T13:54:38Z","snapshot_observed_at":"2026-08-22T14:10:05.229038Z","submitted_at":"2026-04-17T13:54:38Z","title":"TableSeq: Unified Generation of Structure, Content, and Layout","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-10T08:53:14.886099Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2604.16070"},"observation_digest":"sha256:005028b91406e146dce9a2762de58bfed8d7ac974ab9580f636d62373a82cf88","observation_id":"ce28198a-e3fa-4bf3-8ae9-9482a216de9a","resolution":{"observed_at":"2026-05-17T20:50:57.985910Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":"2409.01704","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-07-08T20:35:34.407064Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","venue":"cs.CV","work_id":"67b1592f-02b6-4056-b3fe-cc594e484192","year":2024},"citing_paper":{"arxiv_id":"2605.09343","last_updated":"2026-05-10T05:43:58Z","snapshot_observed_at":"2026-08-17T19:44:08.270539Z","submitted_at":"2026-05-10T05:43:58Z","title":"SKG-VLA: Scene Knowledge Graph Priors for Structured Scene Semantics and Multimodal Reasoning for Decision Making","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-12T04:05:18.970086Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2605.09343"},"observation_digest":"sha256:5a94e1ede0ccdb5840241085de6caa852f1d0e82b95facd887c7eba4e4051972","observation_id":"792ceaf0-50e4-49f6-bf8f-d11706e504e8","resolution":{"observed_at":"2026-05-17T20:50:57.985910Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":"2409.01704","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-07-08T20:35:34.407064Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","venue":"cs.CV","work_id":"67b1592f-02b6-4056-b3fe-cc594e484192","year":2024},"citing_paper":{"arxiv_id":"2605.12623","last_updated":"2026-05-21T05:33:02Z","snapshot_observed_at":"2026-08-13T05:17:47.347775Z","submitted_at":"2026-05-12T18:09:38Z","title":"DocAtlas: Multilingual Document Understanding Across 80+ Languages","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-05-14T21:02:45.148167Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2605.12623"},"observation_digest":"sha256:e9163d0aa90203aa311c8c4d8fdc285caf75c60d504a2eb193ab935a1c6d749a","observation_id":"66b6a92f-8620-4852-b73c-b39f129ee88d","resolution":{"observed_at":"2026-05-17T20:50:57.985910Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":"2409.01704","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-07-08T20:35:34.407064Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","venue":"cs.CV","work_id":"67b1592f-02b6-4056-b3fe-cc594e484192","year":2024},"citing_paper":{"arxiv_id":"2605.12623","last_updated":"2026-05-21T05:33:02Z","snapshot_observed_at":"2026-08-13T05:17:47.347775Z","submitted_at":"2026-05-12T18:09:38Z","title":"DocAtlas: Multilingual Document Understanding Across 80+ Languages","version":2},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-05-22T09:51:40.160096Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2605.12623"},"observation_digest":"sha256:9a4835d06626f7b52bd8b54586d27a36f1123d5d2064e39a3857799b223fef62","observation_id":"44a9747f-c778-4d3a-90e6-0df217f46c00","resolution":{"observed_at":"2026-05-22T09:54:47.097063Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":"2409.01704","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-07-08T20:35:34.407064Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","venue":"cs.CV","work_id":"67b1592f-02b6-4056-b3fe-cc594e484192","year":2024},"citing_paper":{"arxiv_id":"2605.17447","last_updated":"2026-05-17T13:39:47Z","snapshot_observed_at":"2026-08-14T21:46:36.376420Z","submitted_at":"2026-05-17T13:39:47Z","title":"FastOCR: Dynamic Visual Fixation via KV Cache Pruning for Efficient Document Parsing","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-20T14:53:57.376715Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2605.17447"},"observation_digest":"sha256:0ef3524b097d31ff07b44763aed01f17cf85654cf889b6fa6466e7ae585ee1d7","observation_id":"b19bc4fb-50a4-444a-93ed-d0157391db56","resolution":{"observed_at":"2026-05-20T14:58:24.813176Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":"2409.01704","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-07-08T20:35:34.407064Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","venue":"cs.CV","work_id":"67b1592f-02b6-4056-b3fe-cc594e484192","year":2024},"citing_paper":{"arxiv_id":"2605.19866","last_updated":"2026-05-19T13:58:24Z","snapshot_observed_at":"2026-08-16T07:05:25.316498Z","submitted_at":"2026-05-19T13:58:24Z","title":"Structured Layout Priors for Robust Out-of-Distribution Visual Document Understanding","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-20T05:48:34.771799Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2605.19866"},"observation_digest":"sha256:358b7f24af2df4faf0958bacd8e04142364f59a1a7807fa5d388d757fb361bc7","observation_id":"9151640c-0f97-481a-b94c-4225f20be6b8","resolution":{"observed_at":"2026-05-20T05:53:05.057867Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":"2409.01704","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-07-08T20:35:34.407064Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","venue":"cs.CV","work_id":"67b1592f-02b6-4056-b3fe-cc594e484192","year":2024},"citing_paper":{"arxiv_id":"2605.21611","last_updated":"2026-05-20T18:17:50Z","snapshot_observed_at":"2026-08-17T02:34:31.797934Z","submitted_at":"2026-05-20T18:17:50Z","title":"UniVL: Unified Vision-Language Embedding for Spatially Grounded Contextual Image Generation","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-22T09:25:19.066598Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2605.21611"},"observation_digest":"sha256:1c2d936eeeaeee45eedcbcb1007a93432b4be2ce2941a520dc5a3046fc54a57c","observation_id":"53fcbf81-7705-44f2-bdd4-9d5f88fa8717","resolution":{"observed_at":"2026-05-22T09:26:20.695715Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":"2409.01704","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-07-08T20:35:34.407064Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","venue":"cs.CV","work_id":"67b1592f-02b6-4056-b3fe-cc594e484192","year":2024},"citing_paper":{"arxiv_id":"2605.22100","last_updated":"2026-05-28T08:19:59Z","snapshot_observed_at":"2026-08-16T14:14:23.008482Z","submitted_at":"2026-05-21T07:36:41Z","title":"MPDocBench-Parse: Benchmarking Practical Multi-page Document Parsing","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-22T05:58:04.055855Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2605.22100"},"observation_digest":"sha256:9df2afc820acecd4da83f846bef37bd1e8c6c940bf8f5977df698ce5aa196384","observation_id":"855fc805-dc42-4dfc-af08-c65114863b62","resolution":{"observed_at":"2026-05-22T06:01:08.919944Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":"2409.01704","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-07-08T20:35:34.407064Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","venue":"cs.CV","work_id":"67b1592f-02b6-4056-b3fe-cc594e484192","year":2024},"citing_paper":{"arxiv_id":"2605.22100","last_updated":"2026-05-28T08:19:59Z","snapshot_observed_at":"2026-08-16T14:14:23.008482Z","submitted_at":"2026-05-21T07:36:41Z","title":"MPDocBench-Parse: Benchmarking Practical Multi-page Document Parsing","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-06-30T17:37:33.750306Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2605.22100"},"observation_digest":"sha256:7c02a83a0a3fee6db752cfeabcdc4484c8ee0278a239f57067ffbf53e279e1e7","observation_id":"917c201c-0b89-4539-84c8-18eb88a4b3ee","resolution":{"observed_at":"2026-07-01T15:05:48.197561Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":"2409.01704","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-07-08T20:35:34.407064Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","venue":"cs.CV","work_id":"67b1592f-02b6-4056-b3fe-cc594e484192","year":2024},"citing_paper":{"arxiv_id":"2605.22413","last_updated":"2026-05-21T12:37:03Z","snapshot_observed_at":"2026-08-18T12:39:42.314681Z","submitted_at":"2026-05-21T12:37:03Z","title":"From Recognition to Reasoning: Benchmarking and Enhancing MLLMs on Real-World Receipt Document Understanding","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-22T07:36:53.347112Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2605.22413"},"observation_digest":"sha256:0045bb532a1e142397e5bad017a9f8cd63a6e68ff0162a512b1ace3836c0e0d5","observation_id":"4197ac5c-f05a-4f4f-bcdc-448003cc3519","resolution":{"observed_at":"2026-05-22T07:41:15.193383Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":"2409.01704","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-07-08T20:35:34.407064Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","venue":"cs.CV","work_id":"67b1592f-02b6-4056-b3fe-cc594e484192","year":2024},"citing_paper":{"arxiv_id":"2605.27978","last_updated":"2026-05-27T05:16:21Z","snapshot_observed_at":"2026-08-13T01:21:58.936928Z","submitted_at":"2026-05-27T05:16:21Z","title":"ABot-OCR Technical Report","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-06-29T13:29:17.221676Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2605.27978"},"observation_digest":"sha256:e3bdb159ad7c8a5acd9589e62fe6a39530a3db6cdde54a514749f4db9bf5fcca","observation_id":"a817441c-13b4-4a0f-bbfd-6446ee886f0d","resolution":{"observed_at":"2026-06-29T13:33:28.133191Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":"2409.01704","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-07-08T20:35:34.407064Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","venue":"cs.CV","work_id":"67b1592f-02b6-4056-b3fe-cc594e484192","year":2024},"citing_paper":{"arxiv_id":"2606.03264","last_updated":"2026-06-02T07:27:03Z","snapshot_observed_at":"2026-08-16T07:12:09.561259Z","submitted_at":"2026-06-02T07:27:03Z","title":"PaddleOCR-VL-1.6: Expanding the Frontier of Document Parsing with Under-Optimized Region Refinement and Progressive Post-Training","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-06-28T10:39:14.444486Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2606.03264"},"observation_digest":"sha256:1e9d24dcf8c2dcc1186ce42b2f7dd235cea40a21fdcf16c3388caed2be7de61b","observation_id":"58049c56-4317-4cd3-a67c-316ef1a20f74","resolution":{"observed_at":"2026-07-02T02:46:28.805474Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":"2409.01704","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-07-08T20:35:34.407064Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","venue":"cs.CV","work_id":"67b1592f-02b6-4056-b3fe-cc594e484192","year":2024},"citing_paper":{"arxiv_id":"2606.08979","last_updated":"2026-06-08T03:25:20Z","snapshot_observed_at":"2026-08-17T19:50:26.208307Z","submitted_at":"2026-06-08T03:25:20Z","title":"EviProp: Seeded Relevance Diffusion on Chunk-Page Graphs for Long Multimodal Document Retrieval","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-06-27T15:04:30.757297Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2606.08979"},"observation_digest":"sha256:2f11c62b7ca26bc44fd69cfdf2a6a497b62530c7723673e7344443a65555f85e","observation_id":"d0e80d9f-7c30-4861-8632-e3c0ee8ead87","resolution":{"observed_at":"2026-07-03T03:37:35.702362Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":"2409.01704","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-07-08T20:35:34.407064Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","venue":"cs.CV","work_id":"67b1592f-02b6-4056-b3fe-cc594e484192","year":2024},"citing_paper":{"arxiv_id":"2606.15932","last_updated":"2026-06-16T15:28:03Z","snapshot_observed_at":"2026-07-06T23:52:37.922109Z","submitted_at":"2026-06-14T17:21:43Z","title":"Beyond NL2Code: A Structured Survey of Multimodal Code Intelligence","version":2},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-06-27T03:57:19.028507Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2606.15932"},"observation_digest":"sha256:5f8840031f74ced59e2bdf2b8f5d06c332e344a0e6df1c41fc863fb347e9bf58","observation_id":"358b28e6-6d8d-4f1f-9519-50d00677e258","resolution":{"observed_at":"2026-07-03T17:38:44.217382Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":"2409.01704","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-07-08T20:35:34.407064Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","venue":"cs.CV","work_id":"67b1592f-02b6-4056-b3fe-cc594e484192","year":2024},"citing_paper":{"arxiv_id":"2606.23050","last_updated":"2026-06-22T09:01:29Z","snapshot_observed_at":"2026-08-13T17:57:13.876850Z","submitted_at":"2026-06-22T09:01:29Z","title":"Unlimited OCR Works","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-06-26T09:03:33.539615Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2606.23050"},"observation_digest":"sha256:3495ae86ee56f74ff4d1ece39d8fc4ddd267bc580f785f55b266c3410a900d5a","observation_id":"1c48e528-0420-44d2-ab42-5e5b6fd5aa8f","resolution":{"observed_at":"2026-07-04T10:19:46.939595Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":"2409.01704","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-07-08T20:35:34.407064Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","venue":"cs.CV","work_id":"67b1592f-02b6-4056-b3fe-cc594e484192","year":2024},"citing_paper":{"arxiv_id":"2606.24484","last_updated":"2026-06-23T12:18:50Z","snapshot_observed_at":"2026-08-21T03:02:33.320802Z","submitted_at":"2026-06-23T12:18:50Z","title":"Advancing WordArt-Oriented Scene Text Recognition: Datasets and Methods","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-06-26T01:06:05.381643Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2606.24484"},"observation_digest":"sha256:f3741868e362b4510939fa1220e7ec3a5c200307f37c44696893336aa96f7c68","observation_id":"062bec07-ad84-4c20-9a1d-9c1bfd440594","resolution":{"observed_at":"2026-07-04T15:59:57.268402Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":"2409.01704","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-07-08T20:35:34.407064Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","venue":"cs.CV","work_id":"67b1592f-02b6-4056-b3fe-cc594e484192","year":2024},"citing_paper":{"arxiv_id":"2606.29213","last_updated":"2026-06-28T05:46:05Z","snapshot_observed_at":"2026-08-14T02:12:25.132097Z","submitted_at":"2026-06-28T05:46:05Z","title":"Can OCR-VLMs Read Devanagari? A Stress-Test Benchmark and Post-Correction Study","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-30T07:56:37.081738Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2606.29213"},"observation_digest":"sha256:c534ddc73107907fbbebc11948b57b627ee1787957deb1411b68628b41c87e84","observation_id":"1cf928dc-6511-42c1-beef-edc96c8a0884","resolution":{"observed_at":"2026-06-30T08:04:28.313205Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":"2409.01704","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-07-08T20:35:34.407064Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","venue":"cs.CV","work_id":"67b1592f-02b6-4056-b3fe-cc594e484192","year":2024},"citing_paper":{"arxiv_id":"2606.29905","last_updated":"2026-06-29T07:41:39Z","snapshot_observed_at":"2026-08-18T18:12:22.443994Z","submitted_at":"2026-06-29T07:41:39Z","title":"StrucTab: A Structured Optimization Framework for Table Parsing","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-06-30T06:22:21.694355Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2606.29905"},"observation_digest":"sha256:055b5a28026884e5461dc6de954dcab09acca07197ced5bf822b59e1d4557d1b","observation_id":"d6f72525-8cef-4759-a7c2-9bd351202d7e","resolution":{"observed_at":"2026-06-30T06:24:18.973209Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-07-12T05:51:08.895411Z","title":"Vary: Scaling up the vision vocabulary for large vision-language model","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.02956","last_updated":"2026-07-03T04:59:12Z","snapshot_observed_at":"2026-08-14T18:47:59.009039Z","submitted_at":"2026-07-03T04:59:12Z","title":"MORE: A Multilingual Document Parsing Benchmark and Evaluation","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-07-12T05:51:08.895411Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2607.02956"},"observation_digest":"sha256:d72061d7f606edeac6f3b772f5196ebb3e7a52bb8d27ea38d75bf7ac65a3bef2","observation_id":"847bce4b-b337-464c-9cfe-28ed85053190","resolution":{"observed_at":"2026-07-12T05:51:08.895411Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-07-12T04:17:40.198357Z","title":"arXiv preprint arXiv:2409.01704 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.03184","last_updated":"2026-07-03T10:42:34Z","snapshot_observed_at":"2026-08-05T15:07:18.952856Z","submitted_at":"2026-07-03T10:42:34Z","title":"BVS: Bayesian Visual Search with Multimodal Large Language Model for Fine-grained Perception","version":1},"reference_index":229,"source":"arxiv_source","source_observed_at":"2026-07-12T04:17:40.198357Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2607.03184"},"observation_digest":"sha256:53c26b1b1d5f2df1fe860f1eac78529aa01658b7bf99e4aecef92cb029b74a16","observation_id":"49e1fae3-e195-4641-82b5-f925d3b8703f","resolution":{"observed_at":"2026-07-12T04:17:40.198357Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":"2409.01704","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-07-08T20:35:34.407064Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","venue":"cs.CV","work_id":"67b1592f-02b6-4056-b3fe-cc594e484192","year":2024},"citing_paper":{"arxiv_id":"2607.05927","last_updated":"2026-07-07T07:31:05Z","snapshot_observed_at":"2026-08-18T09:52:32.743493Z","submitted_at":"2026-07-07T07:31:05Z","title":"CMDR: Contextual Multimodal Document Retrieval","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-07-08T20:30:59.126121Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2607.05927"},"observation_digest":"sha256:ef3e1d024b4b2486a08dfb08c9b2da265f1a01dad41e804a34cbdc1ee9054eba","observation_id":"e7f01597-c3ed-4648-809f-25843272ac17","resolution":{"observed_at":"2026-07-08T20:35:34.408404Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-07-14T04:45:32.682508Z","title":"General ocr theory: Towards ocr-2.0 via a unified end-to-end model.arXiv preprint arXiv:2409.01704, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.11562","last_updated":"2026-07-13T13:43:39Z","snapshot_observed_at":"2026-08-20T02:58:41.621792Z","submitted_at":"2026-07-13T13:43:39Z","title":"MonkeyOCRv2: A Visual-Text Foundation Model for Document AI","version":1},"reference_index":111,"source":"pdf_text","source_observed_at":"2026-07-14T04:45:32.682508Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2607.11562"},"observation_digest":"sha256:a5b8ded027f5901d52f9e1d7190e8fab488da98ee04d55455576ec8b03961b20","observation_id":"78530a43-8645-4caf-a68d-4e29b682bdd7","resolution":{"observed_at":"2026-07-14T04:45:32.682508Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-08-02T02:58:32.052229Z","title":"General OCR theory: Towards OCR- 2.0 via a unified end-to-end model.arXiv preprint arXiv:2409.01704, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.14041","last_updated":"2026-07-15T17:12:37Z","snapshot_observed_at":"2026-08-09T09:44:33.027135Z","submitted_at":"2026-07-15T17:12:37Z","title":"Multi-Expert Routing for Multi-Domain Low-Resource OCR: A Manchu Case Study","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-02T02:58:32.052229Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2607.14041"},"observation_digest":"sha256:20d252fe879b34528d5f91cd130316bcf0284029c9cb281ddd93cedc22c38a95","observation_id":"cea1a149-72a3-40d7-8737-1d6faa99ee2c","resolution":{"observed_at":"2026-08-02T02:58:32.052229Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-08-01T17:35:39.974969Z","title":"General ocr theory: To- wards ocr-2.0 via a unified end-to-end model,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.17585","last_updated":"2026-08-12T11:53:54Z","snapshot_observed_at":"2026-08-20T08:17:39.303195Z","submitted_at":"2026-07-20T06:04:08Z","title":"Pixel-Space Diffusion Transformers","version":2},"reference_index":117,"source":"pdf_text","source_observed_at":"2026-08-01T17:35:39.974969Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2607.17585"},"observation_digest":"sha256:f0f1533f6f9fa5dd35862cce393072ecdb94a5f60567338eef764d650b7473b4","observation_id":"ded879e8-3c3d-43d7-9fe5-eeb642491ab0","resolution":{"observed_at":"2026-08-01T17:35:39.974969Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-08-01T05:34:45.864348Z","title":"arXiv preprint arXiv:2409.01704","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.22200","last_updated":"2026-07-24T11:13:44Z","snapshot_observed_at":"2026-08-14T00:58:21.265163Z","submitted_at":"2026-07-24T11:13:44Z","title":"LayoutLite: Token-Level Implicit Layout Analysis for Efficient Document OCR","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-01T05:34:45.864348Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2607.22200"},"observation_digest":"sha256:3cedbe559196a68746bd88f10d16003dc5563324b4382cdee6c8d7de4e5edf29","observation_id":"c6ff5059-e7eb-402e-8a62-6df60cd5ae8c","resolution":{"observed_at":"2026-08-01T05:34:45.864348Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-08-04T19:45:56.984456Z","title":"General ocr theory: Towards ocr-2.0 via a unified end-to-end model.arXiv preprint arXiv:2409.01704, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.01848","last_updated":"2026-08-03T07:56:43Z","snapshot_observed_at":"2026-08-17T08:00:09.907753Z","submitted_at":"2026-08-03T07:56:43Z","title":"Decoupling semantics from vision: A framework for faithful visual-text compression evaluation","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-04T19:45:56.984456Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2608.01848"},"observation_digest":"sha256:eccd88eba6164cd11f53fbd3a2d0a43cce16893d080b81f08db8742e639a7c81","observation_id":"af6a8136-d713-4b21-b2a1-66bd23559e0e","resolution":{"observed_at":"2026-08-04T19:45:56.984456Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-08-15T15:05:39.229824Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.02109","last_updated":"2026-08-03T12:08:56Z","snapshot_observed_at":"2026-08-18T12:38:19.644294Z","submitted_at":"2026-08-03T12:08:56Z","title":"Same Semantics, Different Paths: Self-Improving Alignment for Vision-Text Compression","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-15T15:05:39.229824Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2608.02109"},"observation_digest":"sha256:17201d8657a0a6f928d6f5faea04fe5ff765ed986db6341308316d50cf5def67","observation_id":"21c67e3c-020f-49c5-9493-54af6ee44773","resolution":{"observed_at":"2026-08-15T15:05:39.229824Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2409.01704/citation-record","integrity":"/paper/2409.01704/integrity","json":"/paper/2409.01704/citation-record.json","paper":"/paper/2409.01704"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"https://huggingface.co/datasets/Teklia/CASIA-HWDB2-line (2024) 6","venue":null,"work_id":"efa2f0aa-94bb-4f8a-be9a-ef30d147d703","year":2024},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:d6047b39e6ed915f2099f853b544c398a3d13ab0b3829bfba033263ea426dedc","observation_id":"e42b16ce-f7f9-4e15-9848-74c15374e6b8","resolution":{"observed_at":"2026-05-17T20:50:57.939483Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"https://huggingface.co/datasets/Teklia/IAM-line (2024) 6","venue":null,"work_id":"bade56df-693e-494c-bd89-644ff64c339d","year":2024},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:d9b2dc0486ad1d8032a76dc6c73bd62b06b4843887708c078d4204f618ba7264","observation_id":"a0339d30-0c36-4232-b9d2-d7727883167d","resolution":{"observed_at":"2026-05-17T20:50:57.943760Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"https://huggingface.co/datasets/Teklia/NorHand-v3-line (2024) 6","venue":null,"work_id":"e47f2fad-800e-48cd-a291-6eaddb1fe6fd","year":2024},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:c7481f9354c4fb2304fd24d0beba143b6b8cd91eb353702be4c77afa8cddc1c8","observation_id":"86520682-abb3-41b4-91bf-977a563a714b","resolution":{"observed_at":"2026-05-17T20:50:57.945895Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16609","last_updated":"2023-09-28T17:07:49Z","snapshot_observed_at":"2026-08-20T15:31:01.041088Z","submitted_at":"2023-09-28T17:07:49Z","title":"Qwen Technical Report","version":1},"cited_work":{"arxiv_id":"2309.16609","doi":"10.48550/arxiv.2309.16609","metadata_source":"pith","pith_arxiv_id":"2309.16609","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen Technical Report","venue":"cs.CL","work_id":"bb1fd52f-6b2f-437c-9516-37bdf6eb9be8","year":2023},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"cited_paper":"/paper/2309.16609","citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:8399f78b322b044bb812fe33c40d20349b08396c6f8c9e2c5efb61b8c414ff5d","observation_id":"05cf8137-b9ab-41b8-9bfe-f42bfba1c56f","resolution":{"observed_at":"2026-05-17T20:50:57.867470Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-17T16:38:14.101106+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-17T16:38:14.101106+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.12966","last_updated":"2023-10-13T02:41:28Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-24T17:59:17Z","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","version":3},"cited_work":{"arxiv_id":"2308.12966","doi":"10.48550/arxiv.2308.12966","metadata_source":"pith","pith_arxiv_id":"2308.12966","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","venue":"cs.CV","work_id":"cbc2bb21-b6bb-46c0-80bf-107e195ffe10","year":2023},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"cited_paper":"/paper/2308.12966","citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:08ff75bc6124b603822a66b80ff9116dca92800cf0f603238916df3b3bfb19e2","observation_id":"a8e08e09-0ea6-4040-86fc-2d33094a51cb","resolution":{"observed_at":"2026-05-17T20:50:57.844042Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.13418","last_updated":"2023-08-25T15:03:36Z","snapshot_observed_at":"2026-08-12T03:48:04.422679Z","submitted_at":"2023-08-25T15:03:36Z","title":"Nougat: Neural Optical Understanding for Academic Documents","version":1},"cited_work":{"arxiv_id":"2308.13418","doi":"10.48550/arxiv.2308.13418","metadata_source":"pith","pith_arxiv_id":"2308.13418","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Nougat: Neural Optical Understanding for Academic Documents","venue":"cs.LG","work_id":"26c3b627-7e97-40d7-bab3-020936b8196b","year":2023},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"cited_paper":"/paper/2308.13418","citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:aac8345b48f38902ce35cfff38b33902ecb9a2972885528557ba70548dc48cde","observation_id":"9e922595-aaa7-4d6c-a70d-f7aa170edfc3","resolution":{"observed_at":"2026-05-17T20:50:57.852382Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"ACM Computing Surveys (CSUR) 53(4), 1–35 (2020) 7","venue":null,"work_id":"528ab469-f7b8-4598-bec3-d0e41b53f926","year":2020},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:0b4e623c139a36a52e68e2f5ae9df85f2173771b4031fbbb57e1b3785e12cb08","observation_id":"b77623ec-ee2a-416d-813e-bcf2461ac366","resolution":{"observed_at":"2026-05-17T20:50:57.947757Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09987","last_updated":"2024-04-25T06:31:01Z","snapshot_observed_at":"2026-08-19T16:07:04.987669Z","submitted_at":"2024-04-15T17:58:57Z","title":"OneChart: Purify the Chart Structural Extraction via One Auxiliary Token","version":2},"cited_work":{"arxiv_id":"2404.09987","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.09987","snapshot_observed_at":"2026-06-29T18:33:50.477475Z","title":"arXiv preprint arXiv:2404.09987 (2024) 7, 10","venue":null,"work_id":"aef017f5-46e8-44f2-af7b-02c3afac0ca9","year":2024},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"cited_paper":"/paper/2404.09987","citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:de0e957ae6d6dbb799401ca86f78e3698e416684d5184722e89f813bb372ad31","observation_id":"98246ad0-f46f-404a-b663-842a2a3996c1","resolution":{"observed_at":"2026-05-17T20:50:57.896433Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16821","last_updated":"2024-04-29T20:24:30Z","snapshot_observed_at":"2026-08-17T14:16:52.244007Z","submitted_at":"2024-04-25T17:59:19Z","title":"How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites","version":2},"cited_work":{"arxiv_id":"2404.16821","doi":"10.48550/arxiv.2404.16821","metadata_source":"pith","pith_arxiv_id":"2404.16821","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites","venue":"cs.CV","work_id":"3714835e-c5a6-4d7e-950c-be44670ed9e6","year":2024},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"cited_paper":"/paper/2404.16821","citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:9fe347c80396d158b62824ba37c10db235590ece2a7418d5339e010fb3a1611f","observation_id":"b647b900-f7e5-4a9f-89af-cc4dc5a6b2a8","resolution":{"observed_at":"2026-05-17T20:50:57.910456Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.03144","last_updated":"2021-10-12T13:36:26Z","snapshot_observed_at":"2026-08-20T02:19:25.872846Z","submitted_at":"2021-09-07T15:24:40Z","title":"PP-OCRv2: Bag of Tricks for Ultra Lightweight OCR System","version":2},"cited_work":{"arxiv_id":"2109.03144","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2109.03144","snapshot_observed_at":"2026-07-03T13:48:21.606359Z","title":"arXiv preprint arXiv:2109.03144 (2021) 1, 4, 5","venue":null,"work_id":"36d69ebc-4a5f-4de3-aacc-f0b0fa33eafa","year":2021},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"cited_paper":"/paper/2109.03144","citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:fe89b66922db28b5efc1ce4966c0fbc96a933a6b85a001987e8e70228312dde9","observation_id":"7a415611-44e1-4922-97bd-7dcc1c008b6c","resolution":{"observed_at":"2026-05-17T20:50:57.840233Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: International Conference on Machine Learning (ICML) (2006) 4","venue":null,"work_id":"836361fb-4966-4b6b-aea5-fd46f21d02ad","year":2006},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:f3cf4f55f70c2dd42a541442696b1e71269362b01ad7df9c4cff779ece84a7f4","observation_id":"c1fc507a-0fe6-404a-92e2-08f3b1aa7e14","resolution":{"observed_at":"2026-05-17T20:50:57.949638Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Advances in Neural Information Processing Systems 35, 26418–26431 (2022) 5","venue":null,"work_id":"49adfae1-7d05-44d6-87cb-28f4b285c528","year":2022},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:e7b13ea4c7f508f239c66e3055fe3bc74ca1f74f00dcc0cfd5022dd1da58bbdf","observation_id":"840ff4db-3b79-4c46-8d3e-28ee0e780c4e","resolution":{"observed_at":"2026-05-17T20:50:57.951588Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-16T14:08:19.332089Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":"2403.12895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-07-10T17:07:25.743722Z","title":"arXiv preprint (2024) DOI: 10.48550/ arXiv.2403.12895","venue":"cs.CV","work_id":"89b0a3a7-ff7a-49bf-af40-a8cf7ee77b16","year":2024},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:c85ca053925a983ac8befde6f4df6d455573540f4b7b58f793873bd6a30d6745","observation_id":"29680334-4140-4af1-ae06-56f51675dc11","resolution":{"observed_at":"2026-05-17T20:50:57.864055Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Proceedings of the IEEE 86(11), 2278–2324 (1998) 4","venue":null,"work_id":"69264b81-84bb-483d-8a69-ff980d51582e","year":1998},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:3dc3cb1307fe6ebedb2e56bebf5ea286d40c6a7399614b9af575802c8cc265de","observation_id":"bcd9fa5e-d377-4893-a8f5-6ae4c9d6c675","resolution":{"observed_at":"2026-05-17T20:50:57.953392Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2301.12597","last_updated":"2023-06-15T07:57:29Z","snapshot_observed_at":"2026-08-20T12:00:24.760146Z","submitted_at":"2023-01-30T00:56:51Z","title":"BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models","version":3},"cited_work":{"arxiv_id":"2301.12597","doi":"10.48550/arxiv.2301.12597","metadata_source":"pith","pith_arxiv_id":"2301.12597","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models","venue":"cs.CV","work_id":"63d03f4d-15f4-4583-8286-913c19f02294","year":2023},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"cited_paper":"/paper/2301.12597","citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:0ca921f30ae5c459cd453cf74d92c46c8ef02e40973cbb4b3c503906071104bb","observation_id":"30e3625d-7691-41a1-8edc-50a6f1a28613","resolution":{"observed_at":"2026-05-17T20:50:57.875034Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-07-12T21:49:47.354893+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T21:49:47.354893+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: Proceedings of the AAAI Conference on Artificial Intelligence","venue":null,"work_id":"1ecab656-7bd9-4117-aefa-8f7d6d51b8fe","year":2023},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:575134bfbb5ad50f0ee3af62cb6469db4eb4381630ab13ffd74af9de914695ab","observation_id":"d32a899d-f610-43d9-b37c-1605c776805d","resolution":{"observed_at":"2026-05-17T20:50:57.955229Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: European conference on computer vision","venue":null,"work_id":"5c83d6ae-d234-49c6-ab3c-d71a4af6a816","year":2022},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:c5201e08c255ba93806d93e869516c27556fb454ad90032819b09fa180a6c28e","observation_id":"539bbe91-0bd0-45d7-ae30-623f97fb16ca","resolution":{"observed_at":"2026-05-17T20:50:57.957199Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: Proceedings of the Thirty-First AAAI Conference on Artificial Intelligence (2017) 4","venue":null,"work_id":"d3ba84e7-0a55-400c-af69-f621568f51d0","year":2017},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:a714e1d8a3620b7732141f5ab81d5c218b2755fde743428da2e3cc80333d27f4","observation_id":"0ddbaff4-0cbe-41d7-b9e2-73f17f7aa9e8","resolution":{"observed_at":"2026-05-17T20:50:57.959034Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"IEEE transactions on pattern analysis and machine intelligence 45(1), 919–931 (2022) 4","venue":null,"work_id":"123c9891-67d8-4a1a-9a8c-bdeb3e181dd4","year":2022},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:0317ce5c2d65f814209dc31dada07f793af8edd666fdfc82834d5858bef72c7c","observation_id":"1e7ca359-2f55-4c40-89ea-72244b72c121","resolution":{"observed_at":"2026-05-17T20:50:57.960930Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.14295","last_updated":"2024-05-23T08:15:49Z","snapshot_observed_at":"2026-08-16T13:50:19.823615Z","submitted_at":"2024-05-23T08:15:49Z","title":"Focus Anywhere for Fine-grained Multi-page Document Understanding","version":1},"cited_work":{"arxiv_id":"2405.14295","doi":null,"metadata_source":"pith","pith_arxiv_id":"2405.14295","snapshot_observed_at":"2026-07-08T20:35:34.401926Z","title":"Focus anywhere for fine- grained multi-page document understanding","venue":"cs.CV","work_id":"f9c7794d-c8f0-42dd-b0c1-7bba21f80c07","year":2024},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"cited_paper":"/paper/2405.14295","citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:033cf22385bfbcfde7d59aac195a1938cf60a1c8c83ba7e1bc149ed3702f0b53","observation_id":"816924cd-6ad3-4340-bfc8-13b08c9031ed","resolution":{"observed_at":"2026-05-17T20:50:57.848203Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: Proceedings of the AAAI Conference on Artificial Intelligence","venue":null,"work_id":"8a68b019-a807-4985-a3aa-0dca511aa9ee","year":2024},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:9fbee1a6cf56d90a13a5ea00d85119bc803e79cae9b3ee404d10681ba06c50e1","observation_id":"9a51ff0f-31ed-4f4a-ab7a-2a40e93e16b4","resolution":{"observed_at":"2026-05-17T20:50:57.962772Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.10505","last_updated":"2023-05-23T18:28:39Z","snapshot_observed_at":"2026-08-16T16:07:12.861373Z","submitted_at":"2022-12-20T18:20:50Z","title":"DePlot: One-shot visual language reasoning by plot-to-table translation","version":2},"cited_work":{"arxiv_id":"2212.10505","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2212.10505","snapshot_observed_at":"2026-07-02T02:46:28.845114Z","title":"In: Findings of the 61st Annual Meeting of the Association for Computational Linguistics (2023), https://arxiv.org/abs/ 2212.10505 10","venue":null,"work_id":"5d81d551-cced-4a84-9fda-e5e7e2b9b05d","year":2023},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"cited_paper":"/paper/2212.10505","citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:68079d189415065c4152066a68a95d471cd0ce36d921352ef2d7404c144460e1","observation_id":"83552657-d82b-473b-94ae-9a1f2ec4f7b5","resolution":{"observed_at":"2026-05-17T20:50:57.856462Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"92adcc8d-eb5b-43d5-9310-81b0199b6b9f","year":2024},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:892353c948563a6e9c5b97f18847c9a06da01f03d9a39f6967a8513946246fcf","observation_id":"b005dc02-1751-41b1-9734-0949702479f0","resolution":{"observed_at":"2026-05-17T20:50:57.964571Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"2509bd9d-6de9-45c0-be1b-72de94f05f96","year":2023},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:af548f2f588d5d328c77f6293f9d624a1640f235f3343de3612654d772c2c142","observation_id":"8ea36cf8-f70e-4bfb-aa83-06ddb666f457","resolution":{"observed_at":"2026-05-17T20:50:57.966444Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1912.09641","last_updated":"2019-12-20T04:50:17Z","snapshot_observed_at":"2026-08-13T07:39:45.410432Z","submitted_at":"2019-12-20T04:50:17Z","title":"ICDAR 2019 Robust Reading Challenge on Reading Chinese Text on Signboard","version":1},"cited_work":{"arxiv_id":"1912.09641","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"1912.09641","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"arXiv preprint arXiv:1912.09641 (2019) 8","venue":null,"work_id":"f19d2ec8-ed91-4f3a-972a-3ac0d01e90d7","year":2019},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"cited_paper":"/paper/1912.09641","citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:9c1753dae16ec5f7da24f15857086ec4c5d2d45b81400b89927b244594b2c19c","observation_id":"996bf395-d5f7-471c-8a4d-d67e788c0e1e","resolution":{"observed_at":"2026-05-17T20:50:57.871222Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Pattern Recognition 90, 337–345 (2019) 4","venue":null,"work_id":"05a9fcc5-2331-4f9b-9ccf-808fef52597b","year":2019},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:8f3c6ddf1d09cd01d3527b92b7cba140149cd9661a9796d3e2c63f3aebcaaba7","observation_id":"3016e0fd-e9a8-4736-974f-bfb7a5f3ed67","resolution":{"observed_at":"2026-05-17T20:50:57.968646Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-19T13:06:25.132325Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":"2403.04473","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":"612fe156-781e-49f3-bac5-03fcb13d115f","year":2024},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:6174094d8eac242e2df5aa21710ec937a3df12499b060899e42ab35e7396a80b","observation_id":"3032a2af-cde3-4694-8376-af8e8cfababc","resolution":{"observed_at":"2026-05-17T20:50:57.878737Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1608.03983","last_updated":"2017-05-03T16:28:09Z","snapshot_observed_at":"2026-07-06T05:06:55.589962Z","submitted_at":"2016-08-13T13:46:05Z","title":"SGDR: Stochastic Gradient Descent with Warm Restarts","version":5},"cited_work":{"arxiv_id":"1608.03983","doi":"10.21203/rs.3.rs-7055642/v1","metadata_source":"pith","pith_arxiv_id":"1608.03983","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"SGDR: Stochastic Gradient Descent with Warm Restarts","venue":"cs.LG","work_id":"ad476478-c5ea-495b-a454-168c504bbfcc","year":2016},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"cited_paper":"/paper/1608.03983","citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:29e9f05224b8195d6bcfaece621fb70ea4394704048ce3e5d95f7ac43540e442","observation_id":"643a49af-8a9e-4281-8cbe-89752ba8049a","resolution":{"observed_at":"2026-05-17T20:50:57.885866Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: ICLR (2019) 8","venue":null,"work_id":"7c12c249-e99f-4e46-bf51-7f795b04df53","year":2019},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:45e0f20357531047eb71422f1ebc2d567199967d4f1ab7796b38e6cd65a4050f","observation_id":"6bc7756d-b20e-4751-8133-3a5fbc9c09cb","resolution":{"observed_at":"2026-05-17T20:50:57.970572Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: Proceedings of the IEEE conference on computer vision and pattern recognition","venue":null,"work_id":"1082d7d6-8d3e-41b5-b89b-fac31b95f99c","year":2018},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:7c35a72f5e4878d518e956ee3e4c9d297f272decc2f69ef5b722939ad5ca4eb0","observation_id":"642220e4-7796-4971-8edf-2995b5df6bb7","resolution":{"observed_at":"2026-05-17T20:50:57.972562Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.14761","last_updated":"2023-10-10T23:39:25Z","snapshot_observed_at":"2026-08-22T02:38:23.087889Z","submitted_at":"2023-05-24T06:11:17Z","title":"UniChart: A Universal Vision-language Pretrained Model for Chart Comprehension and Reasoning","version":3},"cited_work":{"arxiv_id":"2305.14761","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.14761","snapshot_observed_at":"2026-06-30T05:34:20.095001Z","title":"UniChart: A universal vision-language pretrained model for chart comprehension and reasoning","venue":null,"work_id":"e82bcfda-a47a-425c-8bf5-87b512a6e320","year":2023},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"cited_paper":"/paper/2305.14761","citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:c14567c442599515b85ca12afa57c9205ecaa5e9e04dd8179a655b8366c4aef8","observation_id":"1f4f5004-0d44-4619-a9cc-8b607dd1f329","resolution":{"observed_at":"2026-05-17T20:50:57.916817Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.10244","last_updated":"2022-03-19T05:00:30Z","snapshot_observed_at":"2026-08-11T03:45:43.920485Z","submitted_at":"2022-03-19T05:00:30Z","title":"ChartQA: A Benchmark for Question Answering about Charts with Visual and Logical Reasoning","version":1},"cited_work":{"arxiv_id":"2203.10244","doi":null,"metadata_source":"pith","pith_arxiv_id":"2203.10244","snapshot_observed_at":"2026-07-04T16:39:57.602645Z","title":"ChartQA: A Benchmark for Question Answering about Charts with Visual and Logical Reasoning","venue":"cs.CL","work_id":"8b49b78c-7e1d-4f57-af10-7c11cd63ff7c","year":2022},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"cited_paper":"/paper/2203.10244","citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:55d7491bc3775cf113df7a565ec9843543f3dba527b7f0f10973c7f38518122b","observation_id":"091fe756-3425-42ad-b4b5-28929c4c0ea6","resolution":{"observed_at":"2026-05-17T20:50:57.919606Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: Proceedings of the IEEE/CVF winter conference on applications of computer vision","venue":null,"work_id":"a8c66591-b057-411c-a226-1fff48a02680","year":2021},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:1869b60fa4373f71bce90c37d3af71978edf35c5ff1d185d5b580685185ccab6","observation_id":"38ed69b7-55e1-43c6-87f4-5ad8e28ab190","resolution":{"observed_at":"2026-05-17T20:50:57.974514Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The PracTEX Journal 1, 1–22 (2007) 7","venue":null,"work_id":"180605f3-11a6-4e73-92d4-6ba401aa44bb","year":2007},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:d4f99bb6f7e39c79ae7fc7525ea0d604da494d512c3b896896473aec0a7c971d","observation_id":"411de043-b84b-436b-b8da-7b1c55a8cdd7","resolution":{"observed_at":"2026-05-17T20:50:57.976588Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: Proceedings of the IEEE/CVF Winter Conference on Applications of Computer Vision","venue":null,"work_id":"4b45b85e-918c-4aa9-a6db-e5e96eaad269","year":2020},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:784dae013a36bb8ca3d11a469952d83900335cd1e6d49052d3f9d5dc9e49f5cf","observation_id":"9f5463a2-2ea0-4362-a738-d7d4373b7914","resolution":{"observed_at":"2026-05-17T20:50:57.978590Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"031d0cec-fa44-407c-9d4d-409e0e53be5b","year":2023},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:b4bb3bda887eeae8ac3e187d45b582c50967a1cd6789865f685a33dd1d21335a","observation_id":"a796079b-98c9-42ff-b422-f72ca288d10a","resolution":{"observed_at":"2026-05-17T20:50:57.980490Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: International conference on machine learning","venue":null,"work_id":"94df44e4-876d-4c76-b709-309a791ee42e","year":2021},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:072e611587df72b260a983c4647268542be9c8ed2260ae40276507cf872a0a9f","observation_id":"4088a510-f470-4d09-a79e-25dcdc8b1e90","resolution":{"observed_at":"2026-05-17T20:50:57.982628Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07596","last_updated":"2024-04-29T09:53:57Z","snapshot_observed_at":"2026-08-17T18:18:19.317583Z","submitted_at":"2024-02-12T11:52:21Z","title":"Sheet Music Transformer: End-To-End Optical Music Recognition Beyond Monophonic Transcription","version":2},"cited_work":{"arxiv_id":"2402.07596","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.07596","snapshot_observed_at":"2026-07-03T00:37:29.668995Z","title":"arXiv preprint arXiv:2402.07596 (2024) 7","venue":null,"work_id":"1d3bc01d-83b0-4ed1-a557-fe8e70fe4788","year":2024},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"cited_paper":"/paper/2402.07596","citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:8fa4da11e0684529db15687941b4afc6f41a4b3bd2a04d6bb77465dcb233f834","observation_id":"4db4e10a-7dea-4de5-92c2-a0cdea1b29ea","resolution":{"observed_at":"2026-05-17T20:50:57.860246Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"International Journal on Document Analysis and Recognition (IJDAR) 26(3), 347–362 (2023) 7","venue":null,"work_id":"845acb46-aa8f-42a0-b457-03763be4aeeb","year":2023},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:30c79556a7c5a9cb30d9a79dab473204568ea0c8b6f3285b213354689a51baef","observation_id":"6aae967e-6667-4ed0-b03b-98b87b0feaa1","resolution":{"observed_at":"2026-05-17T20:50:57.984879Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Advances in Neural Information Processing Systems 35, 25278–25294 (2022) 5","venue":null,"work_id":"6f3096ee-9017-44e4-97a2-1f27a05b9130","year":2022},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:01109f8e049546736824f34caa8f4dba4b918757aacbb5dd8ab3fd822c8370c0","observation_id":"c60d3ebe-c1c5-43df-a439-f73fcd965334","resolution":{"observed_at":"2026-05-17T20:50:57.922062Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: 2017 14th iapr international conference on document analysis and recognition (ICDAR)","venue":null,"work_id":"b015d326-4501-46d1-b056-e919453be339","year":2017},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:d8c6902b4d09a776902e6601ed39b5bc1bed37e2e34fa3254b45b81f171d6786","observation_id":"d2acb293-1f50-487c-9330-e23795783204","resolution":{"observed_at":"2026-05-17T20:50:57.924449Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: Proceedings of the IEEE/CVF conference on computer vision and pattern recognition","venue":null,"work_id":"f15494d1-4067-4e41-9f1d-6bb189ae3b92","year":2019},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:1186596d6ec287fb1b1178aa567bb3e07a2a56c5283d6fb67c324a592b2e3b6e","observation_id":"1d0224b0-2eae-40ab-90ce-cb177cdc53f4","resolution":{"observed_at":"2026-05-17T20:50:57.926597Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: European conference on computer vision","venue":null,"work_id":"ef77aba0-6420-4f24-9974-aa267323fc5b","year":2016},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:bb3d45feff984c4dceb932b776e4ba342050921c8c4fa7f4918f6387878d927c","observation_id":"f7bf3861-6a08-4b4b-85af-c94b8f851f99","resolution":{"observed_at":"2026-05-17T20:50:57.928787Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1601.07140","last_updated":"2016-06-19T23:52:14Z","snapshot_observed_at":"2026-08-14T22:13:00.450823Z","submitted_at":"2016-01-26T19:30:34Z","title":"COCO-Text: Dataset and Benchmark for Text Detection and Recognition in Natural Images","version":2},"cited_work":{"arxiv_id":"1601.07140","doi":null,"metadata_source":"pith","pith_arxiv_id":"1601.07140","snapshot_observed_at":"2026-07-04T16:59:57.589641Z","title":"COCO-Text: Dataset and Benchmark for Text Detection and Recognition in Natural Images","venue":"cs.CV","work_id":"ea6ff5e9-689c-41f0-921f-ad32f5e27198","year":2016},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"cited_paper":"/paper/1601.07140","citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:1b973a624d54a0e03dca0a7fd2f875c21fe3f9dffe6be2f1cc97683f0a17eb3a","observation_id":"891e267d-cfe6-4107-95fd-e32739877243","resolution":{"observed_at":"2026-05-17T20:50:57.882154Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition","venue":null,"work_id":"db8fc6d1-9b10-4b13-b927-0206e0e102eb","year":2020},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:c4ae6acbe44d41f970197d24b49982c4091afd9b80df456944bf6994d497947c","observation_id":"0dd7089c-ef9a-4445-b93e-7d6d6abb44de","resolution":{"observed_at":"2026-05-17T20:50:57.931015Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.06109","last_updated":"2023-12-11T04:26:17Z","snapshot_observed_at":"2026-08-18T09:54:01.946714Z","submitted_at":"2023-12-11T04:26:17Z","title":"Vary: Scaling up the Vision Vocabulary for Large Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2312.06109","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.06109","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Vary: Scaling up the vision vocabulary for large vision-language models","venue":null,"work_id":"76fe758e-793c-44d6-b972-5a5680cf3fc3","year":2023},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"cited_paper":"/paper/2312.06109","citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:645b9ff9b97cb3a5ef2df426e711a27b1e0ae19e877f6ea204ee9a5f2715a9f2","observation_id":"d04d7704-ece2-451a-9042-66889085de75","resolution":{"observed_at":"2026-05-17T20:50:57.889473Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.12503","last_updated":"2024-01-23T05:55:26Z","snapshot_observed_at":"2026-08-19T21:51:46.721086Z","submitted_at":"2024-01-23T05:55:26Z","title":"Small Language Model Meets with Reinforced Vision Vocabulary","version":1},"cited_work":{"arxiv_id":"2401.12503","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.12503","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"arXiv preprint arXiv:2401.12503 (2024) 6, 9","venue":null,"work_id":"6b89ffbc-7c0c-491d-b376-55818b045e86","year":2024},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"cited_paper":"/paper/2401.12503","citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:9176ad384cfa44965cc067e35dd349d0e7e1932514064dfc85e944274f00890e","observation_id":"24c23946-3246-4124-a574-e50af50f122a","resolution":{"observed_at":"2026-05-17T20:50:57.893024Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"f0de7e6e-89af-4092-b36f-58588b13d17e","year":2024},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:3ed77c639feea8d2ec02ddd61a657bc0fe704661f38bf0d8acad0a0e056d37f3","observation_id":"447735a3-1299-4a13-9074-92cfea3d70c4","resolution":{"observed_at":"2026-05-17T20:50:57.933072Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.02499","last_updated":"2023-07-04T11:28:07Z","snapshot_observed_at":"2026-08-18T18:37:49.349707Z","submitted_at":"2023-07-04T11:28:07Z","title":"mPLUG-DocOwl: Modularized Multimodal Large Language Model for Document Understanding","version":1},"cited_work":{"arxiv_id":"2307.02499","doi":"10.48550/arxiv.2307.02499","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.02499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"mplug-docowl: Modularized multimodal large language model for document understanding","venue":"arXiv (Cornell University)","work_id":"990d95b3-f658-488d-b1a6-169e9e6aa7fb","year":2023},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"cited_paper":"/paper/2307.02499","citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:545afdc605bef11edca3f02f9c09513611f69f3bfbb47793f4f748374b035796","observation_id":"ca4339bc-d3f9-4dc1-81b3-4fc30fb7f0ce","resolution":{"observed_at":"2026-05-17T20:50:57.899919Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.05126","last_updated":"2023-10-08T11:33:09Z","snapshot_observed_at":"2026-08-21T09:42:09.574596Z","submitted_at":"2023-10-08T11:33:09Z","title":"UReader: Universal OCR-free Visually-situated Language Understanding with Multimodal Large Language Model","version":1},"cited_work":{"arxiv_id":"2310.05126","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.05126","snapshot_observed_at":"2026-07-03T16:18:37.543028Z","title":"Ureader: Universal ocr-free visually-situated language understanding with multimodal large language model","venue":null,"work_id":"acc4fea5-7f8a-48c1-b1c9-2389be85b798","year":2023},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"cited_paper":"/paper/2310.05126","citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:9188d6d9ceedfba1c1038ec03e9ae64c73d2b3ddd0ace79a8f3e9ffd6b52bd6a","observation_id":"bf3a85f8-994d-4172-92fc-4ad13c8c5fd8","resolution":{"observed_at":"2026-05-17T20:50:57.903141Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1903.10412","last_updated":"2019-03-25T15:52:32Z","snapshot_observed_at":"2026-08-14T16:57:10.305493Z","submitted_at":"2019-03-25T15:52:32Z","title":"ShopSign: a Diverse Scene Text Dataset of Chinese Shop Signs in Street Views","version":1},"cited_work":{"arxiv_id":"1903.10412","doi":null,"metadata_source":"pith","pith_arxiv_id":"1903.10412","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"ShopSign: a Diverse Scene Text Dataset of Chinese Shop Signs in Street Views","venue":"cs.CV","work_id":"4bef364b-5a65-406f-beb9-2a4e570d33db","year":2019},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"cited_paper":"/paper/1903.10412","citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:fecb887dea85fe79e5077d139442f42ab29ddc96044e36672d33e798cf560ad0","observation_id":"bbc42b83-15e6-4f1d-9cb9-a75ed3bc9308","resolution":{"observed_at":"2026-05-17T20:50:57.906823Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: Proceedings of the IEEE/CVF International Conference on Computer Vision","venue":null,"work_id":"bd47d41c-10e3-457d-b3c5-115706ca6b5d","year":2021},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:d90d327ff0e1c15ff2e9a5eac961fa21a4acfd220a96042b65ff25434b4ac4cd","observation_id":"04376c78-3274-44d4-bcf0-6e34456bff3d","resolution":{"observed_at":"2026-05-17T20:50:57.935125Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2205.01068","last_updated":"2022-06-21T17:04:40Z","snapshot_observed_at":"2026-08-06T03:13:37.403059Z","submitted_at":"2022-05-02T17:49:50Z","title":"OPT: Open Pre-trained Transformer Language Models","version":4},"cited_work":{"arxiv_id":"2205.01068","doi":"10.48550/arxiv.2205.01068","metadata_source":"pith","pith_arxiv_id":"2205.01068","snapshot_observed_at":"2026-07-11T03:37:45.880117Z","title":"OPT: Open Pre-trained Transformer Language Models","venue":"cs.CL","work_id":"d7ff3b21-1fff-4cf4-952a-4714e3ef2307","year":2022},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"cited_paper":"/paper/2205.01068","citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:d02bc73fe288c9842b9f6402104d01ae41e97ee000b0e6e7b64f4e36649f7344","observation_id":"297a21fe-7022-469c-9b71-a0740b0ddfe9","resolution":{"observed_at":"2026-05-17T20:50:57.913525Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-05-25T10:53:17.026227+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-25T10:53:17.026227+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: 2019 International conference on document analysis and recognition (ICDAR)","venue":null,"work_id":"cb3f69ec-67dc-41e3-8835-88771076a76d","year":2019},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:020e563e8964fbbf38bc1302eee96c98d5fd0453240273c843b17ea505a5bd48","observation_id":"5e999e3a-48c5-49a6-9493-490274c9b75b","resolution":{"observed_at":"2026-05-17T20:50:57.937198Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In: Proceedings of the IEEE International Conference on Computer Vision (ICCV) (2017) 4 19","venue":null,"work_id":"ca348b85-31bf-429c-a071-6a52837ba13b","year":2017},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:3faa2aeb16abe6e8ed0c62160b6b881e38559b8078f509bdaa86e1098e7a367c","observation_id":"ecabadaf-de54-4320-9096-3b59a7570d2f","resolution":{"observed_at":"2026-05-17T20:50:57.941507Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model"},"reference_resolution":{"displayed":55,"state_counts":{"malformed_identifier":0,"metadata_mismatch":16,"parse_uncertain":0,"unresolved":4,"verified_exact":7,"verified_fuzzy":28},"total_outbound_references":55},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"thesis":"As of 24 August 2026, this Paper Citation Record lists 55 of 55 outbound references and 75 inbound Pith citation observations for arXiv:2409.01704."}