{"as_of":"2026-08-19T14:26:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:3a8086d7c897a8c44460a4e650c34a8ccfd6d3d99351e1797924f80305e4d0f3","coverage":[{"denominator":50,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":50,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-11T23:47:10.303241Z","state":"measured"},{"denominator":68,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":68,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-19T06:32:44.657259+00:00","state":"measured"},{"denominator":18,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":18,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T20:31:36.645575Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T05:47:41.904748Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"cited_work":{"arxiv_id":"2412.02210","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.02210","snapshot_observed_at":"2026-07-03T05:47:41.904748Z","title":"Cc-ocr: A comprehensive and challenging ocr benchmark for evaluating large multimodal models in literacy","venue":null,"work_id":"904d3939-c480-4ddf-a962-a461501704de","year":2024},"citing_paper":{"arxiv_id":"2501.00321","last_updated":"2025-06-05T02:59:05Z","snapshot_observed_at":"2026-08-12T17:21:52.298102Z","submitted_at":"2024-12-31T07:32:35Z","title":"OCRBench v2: An Improved Benchmark for Evaluating Large Multimodal Models on Visual Text Localization and Reasoning","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-17T20:33:26.613927Z"},"links":{"cited_paper":"/paper/2412.02210","citing_paper":"/paper/2501.00321"},"observation_digest":"sha256:39c1cac3db07ea16782340ef10f435c66a49b0f045d65696488a813d8c80f990","observation_id":"44a67d08-3b66-41b7-a987-03b28e41f3aa","resolution":{"observed_at":"2026-05-17T20:33:26.764403Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.02210","snapshot_observed_at":"2026-08-15T20:31:36.645575Z","title":"Cc-ocr: A comprehensive and challenging ocr benchmark for evaluating large multimodal models in literacy","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.12766","last_updated":"2025-05-19T06:45:18Z","snapshot_observed_at":"2026-08-19T13:06:54.634082Z","submitted_at":"2025-05-19T06:45:18Z","title":"Reasoning-OCR: Can Large Multimodal Models Solve Complex Logical Reasoning Problems from OCR Cues?","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-15T20:31:36.645575Z"},"links":{"cited_paper":"/paper/2412.02210","citing_paper":"/paper/2505.12766"},"observation_digest":"sha256:a3734a79ee82088bd5e57030bbcbb1684bc4f26974a01222ea56a9e04fcc0b81","observation_id":"1e8e66b1-1f2e-478f-b609-8a51f5b3f20c","resolution":{"observed_at":"2026-08-15T20:31:36.645575Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.02210","snapshot_observed_at":"2026-08-07T14:57:21.254510Z","title":"Cc-ocr: A comprehensive and challenging ocr benchmark for evaluating large multimodal models in literacy.arXiv preprint arXiv:2412.02210, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17163","last_updated":"2026-05-26T01:50:12Z","snapshot_observed_at":"2026-08-13T12:30:50.129007Z","submitted_at":"2025-05-22T15:25:14Z","title":"OCR-Reasoning Benchmark: Unveiling the True Capabilities of MLLMs in Complex Text-Rich Image Reasoning","version":2},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-07T14:57:21.254510Z"},"links":{"cited_paper":"/paper/2412.02210","citing_paper":"/paper/2505.17163"},"observation_digest":"sha256:6a2aa13010cc27adeeebceeb065c267d8e50ba74f1fac7d57f9856d77f546915","observation_id":"736416f8-5d2a-4a1a-8b65-af190c10210b","resolution":{"observed_at":"2026-08-07T14:57:21.254510Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.02210","snapshot_observed_at":"2026-08-07T13:30:21.170204Z","title":"Cc-ocr: A comprehensive and challenging ocr benchmark for evaluating large multimodal models in literacy.arXiv preprint arXiv:2412.02210, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.21771","last_updated":"2026-05-23T17:30:43Z","snapshot_observed_at":"2026-08-16T14:24:45.654182Z","submitted_at":"2025-05-27T21:09:11Z","title":"MMTABREAL: Real-World Benchmark for Multimodal Table Understanding","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T13:30:21.170204Z"},"links":{"cited_paper":"/paper/2412.02210","citing_paper":"/paper/2505.21771"},"observation_digest":"sha256:1fb7bee37c0f846a1cb5933ba7cf78778b7727f5beabf326efb23bd6ad5e2c48","observation_id":"2f7b4b42-248e-43ad-ab69-7b0e71a560c5","resolution":{"observed_at":"2026-08-07T13:30:21.170204Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.02210","snapshot_observed_at":"2026-08-07T04:09:38.470496Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.11820","last_updated":"2025-06-13T14:23:38Z","snapshot_observed_at":"2026-08-16T03:41:24.518662Z","submitted_at":"2025-06-13T14:23:38Z","title":"Rethinking Multilingual Vision-Language Translation: Dataset, Evaluation, and Adaptation","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T04:09:38.470496Z"},"links":{"cited_paper":"/paper/2412.02210","citing_paper":"/paper/2506.11820"},"observation_digest":"sha256:8d73d4015f95c80b32ed2b1b370fdf139b7343260f9cfd278a6e2580f1c98fbf","observation_id":"c4292ea1-2568-4750-ae9c-1109fad454ab","resolution":{"observed_at":"2026-08-07T04:09:38.470496Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.02210","snapshot_observed_at":"2026-08-06T15:26:02.315456Z","title":"Cc-ocr: A comprehensive and challenging ocr benchmark for evaluating large multimodal models in literacy","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.16863","last_updated":"2026-06-24T02:13:52Z","snapshot_observed_at":"2026-08-18T02:28:59.349533Z","submitted_at":"2025-07-21T21:50:16Z","title":"Position: Reasoning After Perception Means Reasoning Without Vision","version":2},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-06T15:26:02.315456Z"},"links":{"cited_paper":"/paper/2412.02210","citing_paper":"/paper/2507.16863"},"observation_digest":"sha256:0d260dfb5ce73fd393a6de58379067ac872e4a4a886b27d0f6eae2588c2e74c2","observation_id":"964a464a-ceb5-4326-8e21-cf34afe55792","resolution":{"observed_at":"2026-08-06T15:26:02.315456Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.02210","snapshot_observed_at":"2026-08-05T23:03:10.162023Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.06009","last_updated":"2025-08-08T04:39:16Z","snapshot_observed_at":"2026-08-17T15:18:27.850836Z","submitted_at":"2025-08-08T04:39:16Z","title":"MathReal: We Keep It Real! A Real Scene Benchmark for Evaluating Math Reasoning in Multimodal Large Language Models","version":1},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-08-05T23:03:10.162023Z"},"links":{"cited_paper":"/paper/2412.02210","citing_paper":"/paper/2508.06009"},"observation_digest":"sha256:afed9aa3011db91d13f842649e08229187d7e58173cfa62bbc601364041dc8fc","observation_id":"0b2729ac-0c6d-42a0-a2a7-401f20bcb0e0","resolution":{"observed_at":"2026-08-05T23:03:10.162023Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"cited_work":{"arxiv_id":"2412.02210","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.02210","snapshot_observed_at":"2026-07-03T05:47:41.904748Z","title":"Cc-ocr: A comprehensive and challenging ocr benchmark for evaluating large multimodal models in literacy","venue":null,"work_id":"904d3939-c480-4ddf-a962-a461501704de","year":2024},"citing_paper":{"arxiv_id":"2508.10016","last_updated":"2026-05-22T12:46:58Z","snapshot_observed_at":"2026-08-12T14:43:37.008792Z","submitted_at":"2025-08-06T16:17:29Z","title":"Training-Free Multimodal Large Language Model Orchestration","version":3},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-05-19T00:12:39.834892Z"},"links":{"cited_paper":"/paper/2412.02210","citing_paper":"/paper/2508.10016"},"observation_digest":"sha256:923182b343db1cde5140b050c4adef9a4e8de5959b850cd4b6e90e1473c92284","observation_id":"79d6a524-fbf5-4e88-97de-bc2ff4c5d8ac","resolution":{"observed_at":"2026-05-19T00:12:54.173395Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"cited_work":{"arxiv_id":"2412.02210","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.02210","snapshot_observed_at":"2026-07-03T05:47:41.904748Z","title":"Cc-ocr: A comprehensive and challenging ocr benchmark for evaluating large multimodal models in literacy","venue":null,"work_id":"904d3939-c480-4ddf-a962-a461501704de","year":2024},"citing_paper":{"arxiv_id":"2508.10016","last_updated":"2026-05-22T12:46:58Z","snapshot_observed_at":"2026-08-12T14:43:37.008792Z","submitted_at":"2025-08-06T16:17:29Z","title":"Training-Free Multimodal Large Language Model Orchestration","version":4},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-05-25T08:02:15.950975Z"},"links":{"cited_paper":"/paper/2412.02210","citing_paper":"/paper/2508.10016"},"observation_digest":"sha256:8f83cdb49ee42eecd94cc0296888f75ca0d78211621978137079f4f50e6b2e9a","observation_id":"4a312208-d36e-4512-bc46-8e0e4077c625","resolution":{"observed_at":"2026-05-25T08:05:30.767369Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.02210","snapshot_observed_at":"2026-08-05T10:52:13.943512Z","title":"CC-OCR: A Comprehensive and Chal- lenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.03615","last_updated":"2025-09-03T18:08:41Z","snapshot_observed_at":"2026-08-17T10:54:25.988526Z","submitted_at":"2025-09-03T18:08:41Z","title":"E-ARMOR: Edge case Assessment and Review of Multilingual Optical Character Recognition","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-05T10:52:13.943512Z"},"links":{"cited_paper":"/paper/2412.02210","citing_paper":"/paper/2509.03615"},"observation_digest":"sha256:b0c806d27a7915b485143707a30eb436b32a7f6ed25a2b81ad158caa60a0c328","observation_id":"8875ee5a-2409-47cc-b173-7ab2f4a3f511","resolution":{"observed_at":"2026-08-05T10:52:13.943512Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"cited_work":{"arxiv_id":"2412.02210","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.02210","snapshot_observed_at":"2026-07-03T05:47:41.904748Z","title":"Cc-ocr: A comprehensive and challenging ocr benchmark for evaluating large multimodal models in literacy","venue":null,"work_id":"904d3939-c480-4ddf-a962-a461501704de","year":2024},"citing_paper":{"arxiv_id":"2509.22186","last_updated":"2025-09-29T16:41:28Z","snapshot_observed_at":"2026-08-15T07:34:12.195990Z","submitted_at":"2025-09-26T10:45:48Z","title":"MinerU2.5: A Decoupled Vision-Language Model for Efficient High-Resolution Document Parsing","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-17T13:25:31.884175Z"},"links":{"cited_paper":"/paper/2412.02210","citing_paper":"/paper/2509.22186"},"observation_digest":"sha256:003f18dbcdeb256d9941c264c810695e50a20e9158119ce8e9c4e39173b3582e","observation_id":"5740e854-0131-47fa-9469-3996b278d76e","resolution":{"observed_at":"2026-05-17T13:25:32.078836Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.02210","snapshot_observed_at":"2026-08-03T22:36:09.992374Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2511.10055","last_updated":"2026-06-02T09:56:13Z","snapshot_observed_at":"2026-08-15T10:07:37.505761Z","submitted_at":"2025-11-13T07:57:10Z","title":"Physical Plausibility Reasoning via HCM-GRPO: Empowering Compact Model for Superior Performance","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-03T22:36:09.992374Z"},"links":{"cited_paper":"/paper/2412.02210","citing_paper":"/paper/2511.10055"},"observation_digest":"sha256:67118b1902eb8592491ba418b370ef3640efca93805b355815a4fe0a498d9d11","observation_id":"5ddc59e4-c439-4678-8284-13ebfcc68c12","resolution":{"observed_at":"2026-08-03T22:36:09.992374Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"cited_work":{"arxiv_id":"2412.02210","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.02210","snapshot_observed_at":"2026-07-03T05:47:41.904748Z","title":"Cc-ocr: A comprehensive and challenging ocr benchmark for evaluating large multimodal models in literacy","venue":null,"work_id":"904d3939-c480-4ddf-a962-a461501704de","year":2024},"citing_paper":{"arxiv_id":"2511.14998","last_updated":"2026-04-07T03:13:19Z","snapshot_observed_at":"2026-08-15T01:12:11.347044Z","submitted_at":"2025-11-19T00:41:14Z","title":"FinCriticalED: A Visual Benchmark for Financial Fact-Level OCR","version":3},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-17T20:19:52.701262Z"},"links":{"cited_paper":"/paper/2412.02210","citing_paper":"/paper/2511.14998"},"observation_digest":"sha256:c5a2f5fec9fa7f38ac653c09566857836ede89977ff8633fecf01c7bdf64b810","observation_id":"cfe939af-8412-4dfb-ab86-9f248e99c88e","resolution":{"observed_at":"2026-05-17T20:20:11.506437Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"cited_work":{"arxiv_id":"2412.02210","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.02210","snapshot_observed_at":"2026-07-03T05:47:41.904748Z","title":"Cc-ocr: A comprehensive and challenging ocr benchmark for evaluating large multimodal models in literacy","venue":null,"work_id":"904d3939-c480-4ddf-a962-a461501704de","year":2024},"citing_paper":{"arxiv_id":"2603.23885","last_updated":"2026-04-18T12:13:54Z","snapshot_observed_at":"2026-08-11T00:45:34.952402Z","submitted_at":"2026-03-25T03:19:09Z","title":"Towards Real-World Document Parsing via Realistic Scene Synthesis and Document-Aware Training","version":3},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-15T01:15:26.757215Z"},"links":{"cited_paper":"/paper/2412.02210","citing_paper":"/paper/2603.23885"},"observation_digest":"sha256:4bfe6c2410579b38788897b4e50bd31b4a4948693155404913a8ea2ff8f3e601","observation_id":"a18699c5-30c6-40c8-910e-6ffe2189aff8","resolution":{"observed_at":"2026-05-15T01:18:26.614884Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"cited_work":{"arxiv_id":"2412.02210","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.02210","snapshot_observed_at":"2026-07-03T05:47:41.904748Z","title":"Cc-ocr: A comprehensive and challenging ocr benchmark for evaluating large multimodal models in literacy","venue":null,"work_id":"904d3939-c480-4ddf-a962-a461501704de","year":2024},"citing_paper":{"arxiv_id":"2605.11301","last_updated":"2026-05-11T22:42:12Z","snapshot_observed_at":"2026-08-15T01:08:53.505050Z","submitted_at":"2026-05-11T22:42:12Z","title":"LatentRouter: Can We Choose the Right Multimodal Model Before Seeing Its Answer?","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-13T01:42:54.802658Z"},"links":{"cited_paper":"/paper/2412.02210","citing_paper":"/paper/2605.11301"},"observation_digest":"sha256:8bcf1492e4ec6cd34a50d0172dc7f5f8ea4ea88a58d6bb5d0776bde62693bbd0","observation_id":"947b5532-e2b7-4993-a8c3-a44d4155f191","resolution":{"observed_at":"2026-05-13T01:47:04.327801Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"cited_work":{"arxiv_id":"2412.02210","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.02210","snapshot_observed_at":"2026-07-03T05:47:41.904748Z","title":"Cc-ocr: A comprehensive and challenging ocr benchmark for evaluating large multimodal models in literacy","venue":null,"work_id":"904d3939-c480-4ddf-a962-a461501704de","year":2024},"citing_paper":{"arxiv_id":"2605.26712","last_updated":"2026-05-26T08:53:34Z","snapshot_observed_at":"2026-07-06T23:36:29.930024Z","submitted_at":"2026-05-26T08:53:34Z","title":"METATR: A Multilingual, Evolving Benchmark for Automatic Text Recognition","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-06-29T18:48:45.623530Z"},"links":{"cited_paper":"/paper/2412.02210","citing_paper":"/paper/2605.26712"},"observation_digest":"sha256:7c51235b08c70b548e9b45289e80995f49bf163aa0e93915e0a33bef42ec5244","observation_id":"5a521e15-f43c-4ae3-b776-ccd1164a5765","resolution":{"observed_at":"2026-06-29T18:53:51.525916Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"cited_work":{"arxiv_id":"2412.02210","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2412.02210","snapshot_observed_at":"2026-07-03T05:47:41.904748Z","title":"Cc-ocr: A comprehensive and challenging ocr benchmark for evaluating large multimodal models in literacy","venue":null,"work_id":"904d3939-c480-4ddf-a962-a461501704de","year":2024},"citing_paper":{"arxiv_id":"2606.11477","last_updated":"2026-06-09T22:12:43Z","snapshot_observed_at":"2026-08-15T04:26:31.935397Z","submitted_at":"2026-06-09T22:12:43Z","title":"Towards Fully Automated Exam Grading: Fairness-Aware Recognition of Handwritten Answers with Foundation Models","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-06-27T13:03:07.941337Z"},"links":{"cited_paper":"/paper/2412.02210","citing_paper":"/paper/2606.11477"},"observation_digest":"sha256:477412a361d7e00180ea00ea16d5292ad0190de7a9bdb4ba4bce5fab04d85aa5","observation_id":"ef1ae605-bc50-4308-b92b-e56086257121","resolution":{"observed_at":"2026-07-03T05:47:41.906627Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.02210","snapshot_observed_at":"2026-07-31T06:20:14.141939Z","title":"CC-OCR: A comprehensive and challenging OCR benchmark for evaluating large multimodal models in literacy.arXiv preprint arXiv:2412.02210, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24904","last_updated":"2026-07-27T17:59:53Z","snapshot_observed_at":"2026-08-15T01:08:05.619745Z","submitted_at":"2026-07-27T17:59:53Z","title":"Mage-VL: An Efficient Codec-Native Streaming Multimodal Foundation Model","version":1},"reference_index":131,"source":"pdf_text","source_observed_at":"2026-07-31T06:20:14.141939Z"},"links":{"cited_paper":"/paper/2412.02210","citing_paper":"/paper/2607.24904"},"observation_digest":"sha256:65b30ba22fc233a05d33da64defae8cf25e7c5392c72062f86bc64e7a044eefc","observation_id":"1aa6dd25-2830-4ce1-8020-ba4edd373b63","resolution":{"observed_at":"2026-07-31T06:20:14.141939Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2412.02210/citation-record","integrity":"/paper/2412.02210/integrity","json":"/paper/2412.02210/citation-record.json","paper":"/paper/2412.02210"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T23:47:11.383201Z","title":null,"venue":null,"work_id":"e51fb093-1336-481f-a9f0-2d6e7094cc8e","year":null},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.019976Z"},"links":{"citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:ecf3fbb06144336fc17fc9f3fa056ea520349a222fde8081671685343e1d9f92","observation_id":"4d312f58-a6a5-4c9c-9810-649d1a422ac7","resolution":{"observed_at":"2026-08-11T23:47:11.388706Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-17T09:58:46.058102Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-11T23:47:10.026585Z","title":"Gpt-4 technical report","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.026585Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:b06a958894cc7c640e4be06704d60e91840443aa4de314ce5222ff37d51c8fbf","observation_id":"31aa5d48-02f9-4e3f-842d-dd6422a2d22f","resolution":{"observed_at":"2026-08-11T23:47:10.026585Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.12966","last_updated":"2023-10-13T02:41:28Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-24T17:59:17Z","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.12966","snapshot_observed_at":"2026-08-11T23:47:10.033823Z","title":"Qwen-vl: A frontier large vision-language model with versatile abilities","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.033823Z"},"links":{"cited_paper":"/paper/2308.12966","citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:c04e7d48907f6e03e36089ad1698d7995a2e86efe2c54a205db66cf880cf715e","observation_id":"2e9d4d9c-600d-42f8-9dad-dcdbd68c47c7","resolution":{"observed_at":"2026-08-11T23:47:10.033823Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.13418","last_updated":"2023-08-25T15:03:36Z","snapshot_observed_at":"2026-08-12T03:48:04.422679Z","submitted_at":"2023-08-25T15:03:36Z","title":"Nougat: Neural Optical Understanding for Academic Documents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.13418","snapshot_observed_at":"2026-08-11T23:47:10.040694Z","title":"Nougat: Neural optical understanding for academic documents","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.040694Z"},"links":{"cited_paper":"/paper/2308.13418","citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:556489779e542431ffc0638b1147b61a845799311364e512be9755c4262bb7a8","observation_id":"1ee5e262-a344-4d7b-9077-902b4abb54ce","resolution":{"observed_at":"2026-08-11T23:47:10.040694Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T23:47:11.362394Z","title":"Onechart: Purify the chart structural extraction via one auxiliary token","venue":null,"work_id":"ef14b90d-1569-41b9-bb16-8187dbdf6055","year":2024},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.047209Z"},"links":{"citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:c6b299ab465bb9b3d4a77763f9fdcd6d7fd284efc1b7e4b84119ca11746bafe8","observation_id":"0ccdb98b-af15-4edf-b688-0495748d2e3c","resolution":{"observed_at":"2026-08-11T23:47:11.368604Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T23:47:11.343853Z","title":"Internvl: Scaling up vision foundation models and aligning for generic visual-linguistic tasks","venue":null,"work_id":"dd3cca9c-79ea-4916-89d0-da89dfe92c7e","year":2024},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.053297Z"},"links":{"citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:60979b7fa37c0e85b446c86dc24f003a92d11a9b31605ec31682026480c53481","observation_id":"85f1cf64-7422-4120-b920-959fd9266711","resolution":{"observed_at":"2026-08-11T23:47:11.350282Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T23:47:11.324431Z","title":"Total-text: A com- prehensive dataset for scene text detection and recognition","venue":null,"work_id":"ab658fd3-b590-41ee-9d1c-f1a3a31ab125","year":2017},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.060223Z"},"links":{"citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:64780ceac7c3937f7a9d25197809b86b09850ca0c57002daef868514c6f1feb2","observation_id":"1b189e01-9fe0-4e59-9b2d-6623439d76d3","resolution":{"observed_at":"2026-08-11T23:47:11.330529Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T23:47:11.302765Z","title":"Icpr2018 contest on robust reading for multi- type web images","venue":null,"work_id":"1c2760bb-be62-4791-9ee3-460af9eeb409","year":2018},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.066507Z"},"links":{"citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:cf9de37e128c8ed05dd7a8b96277fb0c2181954292e0697a172bc3518da5048f","observation_id":"8700ec30-408a-4a54-928e-c9ac28f40d6e","resolution":{"observed_at":"2026-08-11T23:47:11.308637Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-16T14:08:19.332089Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-08-11T23:47:10.073378Z","title":"mplug-docowl 1.5: Unified structure learning for ocr-free document understanding","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.073378Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:f354b87cf5e8e5cd35e1b23d373b15bd1f8648f82022b98d4b88f1b8d63bb6fa","observation_id":"6134128f-727b-48ea-b3de-13e1898d02a4","resolution":{"observed_at":"2026-08-11T23:47:10.073378Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T23:47:11.284086Z","title":null,"venue":null,"work_id":"c08061ab-bea2-4502-a194-7f8f36b924da","year":2019},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.079322Z"},"links":{"citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:2b7036928781afd7df8f455e19e8a6abf98da6da3d0441a6011217a3669356d4","observation_id":"00f847f6-4254-4297-9a4c-3ec2ff245a1e","resolution":{"observed_at":"2026-08-11T23:47:11.290550Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T23:47:11.264703Z","title":"Post-ocr parsing: building simple and robust parser via bio tagging","venue":null,"work_id":"98e7bdea-5791-4fd4-8f78-b8f06448bafc","year":2019},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.085071Z"},"links":{"citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:3e51f9ffe645867c3c2f60abddfaa4c6f4ea03e29956b0e2cdc0a00ee181aa67","observation_id":"9a410f80-8faf-42e5-909b-6e6226005c6d","resolution":{"observed_at":"2026-08-11T23:47:11.271230Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T23:47:11.247610Z","title":"Funsd: A dataset for form understanding in noisy scanned documents","venue":null,"work_id":"534beccd-1dbf-4e86-9847-50255a6be419","year":2019},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.090339Z"},"links":{"citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:81e0bc49bae44dc77c591de1299876bb20b475e0a1cf73ad57bc1302127e878d","observation_id":"15702156-a23a-436d-971a-6e4dfcfcccaf","resolution":{"observed_at":"2026-08-11T23:47:11.253184Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T23:47:11.230335Z","title":"Icdar 2015 competition on robust reading","venue":null,"work_id":"a7a70b1a-b43d-46a8-b57f-1f6307a64986","year":2015},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.095690Z"},"links":{"citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:0599680801d092eb91e65ccbd53c2714a5d363a5e756214de84db489b77bdd65","observation_id":"a83c0578-8d9f-49fc-a64c-e6378e62c762","resolution":{"observed_at":"2026-08-11T23:47:11.236087Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T23:47:11.210074Z","title":"Ocr-free document understanding transformer","venue":null,"work_id":"5c19c553-c143-4ed9-8bf8-32ee96d4c695","year":2022},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.100963Z"},"links":{"citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:9f26735caa6d0acbd0177000dd2bdadb6a3fd64d9783ce93734c912c4e1f2f82","observation_id":"23ba1a9b-30ac-4852-beab-aa6af8349b5c","resolution":{"observed_at":"2026-08-11T23:47:11.216001Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T23:47:11.189547Z","title":"Visual information extraction in the wild: practical dataset and end-to-end solu- tion","venue":null,"work_id":"fde56aa1-3adc-4fb2-9ac9-bf01b5c4a745","year":2023},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.106119Z"},"links":{"citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:48d75cf00da78bdaefbb2d74bbc3e0ce3f0fe115883a742c588900a66cbf8ea9","observation_id":"77421711-b0f8-4c55-b65b-f8a221ac9727","resolution":{"observed_at":"2026-08-11T23:47:11.196174Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T23:47:11.169568Z","title":"Binary coors capable or ‘correcting deletions, insertions, and reversals","venue":null,"work_id":"c31d8b0f-d25c-4909-994a-1a6cc95f8164","year":1966},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.111356Z"},"links":{"citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:99784e45e1eb7f6a729db66bbdd47b0d751e4d18ce36451d61dd4c1fb3ce5872","observation_id":"0fa47a8c-203b-4719-a464-ed83920f3e01","resolution":{"observed_at":"2026-08-11T23:47:11.176016Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T23:47:11.149106Z","title":"TableBank: Table benchmark for image- based table detection and recognition","venue":null,"work_id":"abe1a87f-dd2f-4e54-9b1e-94ff36b40797","year":1918},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.116371Z"},"links":{"citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:4e746efdc58f26996356c6e44cb87dcaf2532d911cdfda7b618d80cdabe91b72","observation_id":"4b8f8499-f103-4abe-b669-406e7323761b","resolution":{"observed_at":"2026-08-11T23:47:11.154803Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.14295","last_updated":"2024-05-23T08:15:49Z","snapshot_observed_at":"2026-08-16T13:50:19.823615Z","submitted_at":"2024-05-23T08:15:49Z","title":"Focus Anywhere for Fine-grained Multi-page Document Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.14295","snapshot_observed_at":"2026-08-11T23:47:10.121280Z","title":"Focus anywhere for fine- grained multi-page document understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.121280Z"},"links":{"cited_paper":"/paper/2405.14295","citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:cf6706a47f15defde12eb1f67c482ba28381c6b93683b531b5fa8e827cbcdd5c","observation_id":"87624657-47b0-4cdc-9cf4-d1c3dea777f0","resolution":{"observed_at":"2026-08-11T23:47:10.121280Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.07895","last_updated":"2024-08-26T02:37:14Z","snapshot_observed_at":"2026-08-13T22:21:37.032118Z","submitted_at":"2023-05-13T11:28:37Z","title":"OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.07895","snapshot_observed_at":"2026-08-11T23:47:10.126191Z","title":"On the hidden mystery of ocr in large multimodal models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.126191Z"},"links":{"cited_paper":"/paper/2305.07895","citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:89b7780e914a0f005bead0fb73bd38dedefe3fd638fa93cdc2d6cec42c4480a6","observation_id":"19a303e8-bf54-415e-a658-ed2a8f03fdf1","resolution":{"observed_at":"2026-08-11T23:47:10.126191Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T23:47:11.130448Z","title":"Spts v2: single-point scene text spotting","venue":null,"work_id":"7c50a301-ba4f-4867-beae-72619c26f698","year":2023},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.131612Z"},"links":{"citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:c1197492d5d341badd2543735d71ddbcdd6678c077f3d78b1df9c2b8efed182b","observation_id":"e5bd558f-d825-4799-b185-fdde556e1d39","resolution":{"observed_at":"2026-08-11T23:47:11.136303Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-19T13:06:25.132325Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-08-11T23:47:10.136168Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.136168Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:fc73e41d78d4c0eaa6982c288561bcc279287a0c3503f82d0906926a2945b53d","observation_id":"76f39e67-cb53-499a-8984-cc1e90f7ec3f","resolution":{"observed_at":"2026-08-11T23:47:10.136168Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T23:47:11.109653Z","title":"Parsing table structures in the wild","venue":null,"work_id":"bd34c637-9914-4f75-bd83-b67d497ab9d7","year":2021},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.141463Z"},"links":{"citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:d8164d9de2eceb205be990f54102f060b60b617a0ad922db34c700f5f47582c2","observation_id":"2a484f35-69bc-460f-9ce5-d3e44ed11fe5","resolution":{"observed_at":"2026-08-11T23:47:11.114773Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T23:47:11.089706Z","title":"Towards end-to-end unified scene text detection and layout analysis","venue":null,"work_id":"4bfe9502-51a5-4110-a611-7e9a3c61543d","year":2022},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.147473Z"},"links":{"citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:c29e9eae96a08631122d1dc7e466298f71859eff87c350372026fa2b330abc03","observation_id":"fa1ea388-b6cb-47d1-8f54-40660dbb9143","resolution":{"observed_at":"2026-08-11T23:47:11.095493Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T23:47:11.068403Z","title":"Layoutllm: Layout instruction tuning with large language models for document understanding","venue":null,"work_id":"2328020d-6ce2-40fa-92d9-2ae3add73c28","year":2024},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.152557Z"},"links":{"citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:024d1518b33baa143c4c00b1b2144fb63d54e65afc57d61ffa51f6fc07dfb601","observation_id":"fd6e2449-2c2f-44fb-93cc-a3bb9f6f235b","resolution":{"observed_at":"2026-08-11T23:47:11.075697Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.11419","last_updated":"2024-08-21T16:54:23Z","snapshot_observed_at":"2026-08-16T14:59:04.213971Z","submitted_at":"2023-09-20T15:50:08Z","title":"KOSMOS-2.5: A Multimodal Literate Model","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.11419","snapshot_observed_at":"2026-08-11T23:47:10.158358Z","title":"Kosmos-2.5: A multimodal literate model","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.158358Z"},"links":{"cited_paper":"/paper/2309.11419","citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:78c182428e8d7c6b4b2780aae46389916fc009711efaea1129a30c80d0371a0c","observation_id":"92df3bc4-0195-4ce9-bdc4-f6412729c6ed","resolution":{"observed_at":"2026-08-11T23:47:10.158358Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T23:47:11.050204Z","title":"Icdar 2019 crohme+ tfd: Competition on recognition of handwritten mathematical expressions and typeset formula detection","venue":null,"work_id":"b1c0fdca-6576-43a1-82c0-1aba4f0cd5e1","year":2019},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.164365Z"},"links":{"citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:6a16ceceec61a70147027d677eed8edf164f33d34c7b005441aee409fe2a3819","observation_id":"18dc72a7-fac3-4fa8-853d-ea813e396e4c","resolution":{"observed_at":"2026-08-11T23:47:11.056054Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T23:47:11.031496Z","title":"The iam-database: an english sentence database for offline handwriting recognition","venue":null,"work_id":"a19b0854-3230-4e8a-8556-aa2ac7df44e3","year":2002},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.170323Z"},"links":{"citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:42cfed72fc88e28ebc867d5cce3975792861ffb7a06e9cc8c1cede17548d5a47","observation_id":"b149ca7d-833d-4d4d-a03b-f1f8b2fd010c","resolution":{"observed_at":"2026-08-11T23:47:11.037443Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T23:47:10.176206Z","title":"Docvqa: A dataset for vqa on document images","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.176206Z"},"links":{"citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:c919d7c54201ec7ec464c41d754570c5565693fcb8cf6d0682cfffb9e07156c8","observation_id":"8abdbd23-e6b1-4282-9347-94fe9b313cca","resolution":{"observed_at":"2026-08-11T23:47:10.176206Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T23:47:10.992439Z","title":"Scene text recognition using higher order language priors","venue":null,"work_id":"23050701-e55a-4a6c-922f-a04cfedba839","year":2012},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.181921Z"},"links":{"citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:f19dc791b8f285970b6ca86bb77df1c1a90c7f0ec4d38bf7706fe76ffcf4beca","observation_id":"769ae03a-6242-4162-92e2-935b755750e9","resolution":{"observed_at":"2026-08-11T23:47:10.999313Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T23:47:10.960485Z","title":"Icdar2019 robust reading challenge on multi-lingual scene text detection and recognition—rrc-mlt-2019","venue":null,"work_id":"b77580a4-edc6-4402-b55e-4f9e5f832a27","year":2019},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.187350Z"},"links":{"citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:e96f2365a7b2e1f02bbe5964f86b7b9791dbbbf4c3e8013b6e12651e922197af","observation_id":"a766abb9-13c1-4085-8164-5ac46b60b34b","resolution":{"observed_at":"2026-08-11T23:47:10.970598Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T23:47:10.932738Z","title":"Cord: a consol- idated receipt dataset for post-ocr parsing","venue":null,"work_id":"92fa5145-8e24-4b2a-acd5-4bed9525ca15","year":2019},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.192956Z"},"links":{"citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:18587b20ec0d289b9a16d474981a02d9d91f87fd1470bddd2a79cc9497a0dd1c","observation_id":"fe6323e2-5fdb-4174-a558-21e2c660c109","resolution":{"observed_at":"2026-08-11T23:47:10.944130Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T23:47:10.198930Z","title":"Laion-5b: An open large-scale dataset for training next gen- eration image-text models","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.198930Z"},"links":{"citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:d6c5321e0489ad7cf6a33421a3d205ef51ede2bc8e2b3b6601d1f634bb10372b","observation_id":"1b6993c0-e907-459c-9b38-e425299e7a76","resolution":{"observed_at":"2026-08-11T23:47:10.198930Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T23:47:10.886581Z","title":"Icdar2017 competition on reading chinese text in the wild (rctw-17)","venue":null,"work_id":"091a9fd5-8225-43e5-9381-8362c579c2c6","year":2017},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.204545Z"},"links":{"citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:9715e37d0c4335472df0166e99b33df7e859625a6b6e9267affca365ffa21e69","observation_id":"0cb5d40d-e1cc-4fcc-ae6a-e02835439dff","resolution":{"observed_at":"2026-08-11T23:47:10.893043Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T23:47:10.865079Z","title":"Towards vqa models that can read","venue":null,"work_id":"b5ca3e3f-0619-4c8c-8767-b41ca95382b5","year":2019},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.211029Z"},"links":{"citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:61ca1efb38039c28a3c990e7fcd9be2f0fbfc4407a8074f59b30c530fff8bb12","observation_id":"5f0525b2-d90d-4968-ab5d-d7a0784162fd","resolution":{"observed_at":"2026-08-11T23:47:10.873328Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T23:47:10.844717Z","title":"Seglink++: Detecting dense and arbitrary- shaped scene text by instance-aware component grouping","venue":null,"work_id":"0b5cf4d3-846c-4ebd-b1ad-553735f69e6e","year":2019},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.216475Z"},"links":{"citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:023e9dea880f2fc48b7defd1f4501ab8f48ce1bcb547add96baf60c9f32c58cc","observation_id":"1ce07450-1c4a-4083-8ba2-6ee9a0e841f8","resolution":{"observed_at":"2026-08-11T23:47:10.850688Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T23:47:10.821964Z","title":"Mtvqa: Benchmarking multilingual text-centric visual question answering, 2024","venue":null,"work_id":"36f64c45-6158-491c-b388-508b1ba7b750","year":2024},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.224158Z"},"links":{"citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:1cf1e885479ff10d2d69396eda5531009d47f331b3efa05fd06db671fca79df3","observation_id":"9378978c-3f0c-47be-9c47-877ffb7c6e69","resolution":{"observed_at":"2026-08-11T23:47:10.831095Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T23:47:10.805451Z","title":"Unifying vision, text, and layout for universal document processing","venue":null,"work_id":"474eeaa6-0c16-419c-a3ec-3674f8df82d1","year":2023},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.230406Z"},"links":{"citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:b166f3d30ecb2772ef7e4d799df10bd5365f7c3fe5ddf62c0decb4aba4a05098","observation_id":"22ba9c20-d406-4224-91c4-1a0f4c486bd3","resolution":{"observed_at":"2026-08-11T23:47:10.810874Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-11T23:47:10.236564Z","title":"Gemini: a family of highly capable multimodal models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.236564Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:acfd8279b469644a6388789e44a893beecf95a688d6704657837a01d276eb7e4","observation_id":"9908f796-e438-499c-aaa8-a800c6c338f2","resolution":{"observed_at":"2026-08-11T23:47:10.236564Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2102.06732","last_updated":"2021-01-24T11:05:24Z","snapshot_observed_at":"2026-08-19T04:06:37.701279Z","submitted_at":"2021-01-24T11:05:24Z","title":"Towards Robust Visual Information Extraction in Real World: New Dataset and Novel Solution","version":1},"cited_work":{"arxiv_id":"2102.06732","doi":null,"metadata_source":"pith","pith_arxiv_id":"2102.06732","snapshot_observed_at":"2026-08-11T23:47:10.471791Z","title":"Towards Robust Visual Information Extraction in Real World: New Dataset and Novel Solution","venue":"cs.CV","work_id":"ff1e6e25-f1cc-4cc5-96fa-a263bc16fb52","year":2021},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.243109Z"},"links":{"cited_paper":"/paper/2102.06732","citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:0b06dcfe7cb0f5703e0aae8f0538750e21eb3e533f895b33cb8acce18ba81035","observation_id":"3feefc3a-554d-41bd-b8ad-2f0155313e24","resolution":{"observed_at":"2026-08-11T23:47:10.480277Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12191","last_updated":"2024-10-03T15:54:49Z","snapshot_observed_at":"2026-08-06T05:35:29.109022Z","submitted_at":"2024-09-18T17:59:32Z","title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12191","snapshot_observed_at":"2026-08-11T23:47:10.248636Z","title":"Qwen2-vl: Enhancing vision-language model’s perception of the world at any resolution","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.248636Z"},"links":{"cited_paper":"/paper/2409.12191","citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:8f86e5391d0361fdd7ee322eb565090a3640a28a54aa27ecd8545ef7241bd2ce","observation_id":"857d57d6-d5bb-44be-91f4-ced6d5d97b6b","resolution":{"observed_at":"2026-08-11T23:47:10.248636Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-18T18:59:18.095701Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.01704","snapshot_observed_at":"2026-08-11T23:47:10.254073Z","title":"General ocr theory: Towards ocr-2.0 via a unified end-to-end model","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.254073Z"},"links":{"cited_paper":"/paper/2409.01704","citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:09c500961218bf735684cbd8e6d210ba30321065cabbb6736749574588661a0e","observation_id":"f1ddf2ba-4fa9-4ea4-a4de-3b59de155505","resolution":{"observed_at":"2026-08-11T23:47:10.254073Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T23:47:10.787319Z","title":"Florence-2: Advancing a unified representation for a variety of vision tasks","venue":null,"work_id":"708a3dd1-a01f-4316-800f-e35c2e22875c","year":2024},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.259956Z"},"links":{"citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:31ca95b80a2609a93e0b1dd4bd6ad0afcd63d9131866cb8d16376772716361a6","observation_id":"1a4e9fe0-786e-4466-94ef-e64971c3a63e","resolution":{"observed_at":"2026-08-11T23:47:10.792916Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T23:47:10.765663Z","title":"Modeling entities as semantic points for visual information extraction in the wild","venue":null,"work_id":"7a3349d2-8822-4cef-b049-b839b5c4d62c","year":null},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.264455Z"},"links":{"citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:acbee5719c3e20198840b4fddd3f8e3ed645075dd777406f0cd8bb52574c03df","observation_id":"13e1739b-7b3f-4fa0-a0e7-d06b2bf93856","resolution":{"observed_at":"2026-08-11T23:47:10.772300Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T23:47:10.747549Z","title":"Dptext-detr: Towards better scene text detection with dynamic points in transformer","venue":null,"work_id":"006c1c43-eb37-49c6-86ff-6e38d89e0984","year":2023},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.270597Z"},"links":{"citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:b3e3b5efc7c8cd90331a03d4524ec3ca18059651944eb8f12fe4eb390032b5a8","observation_id":"21a08f2e-8c81-40b5-a8dc-1ed1bb00eca0","resolution":{"observed_at":"2026-08-11T23:47:10.753010Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T23:47:10.729333Z","title":"Icdar 2023 competition on structured text extraction from visually-rich document images","venue":null,"work_id":"f2bb6f07-d50e-4096-91b4-94ae16f8fc87","year":2023},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.275337Z"},"links":{"citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:7c9bc54045b11346fdf256efbc5e88097cb4ff2e8e68ed54fd21ca6b3b5d7d1a","observation_id":"97bd656e-09d2-49df-b599-6dab9842d4cd","resolution":{"observed_at":"2026-08-11T23:47:10.735243Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T23:47:10.710800Z","title":"Syntax-aware network for handwritten mathematical expression recognition","venue":null,"work_id":"76b06c08-e7b0-4770-8c0e-0830ac892ea8","year":2022},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.280683Z"},"links":{"citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:2a7b481edff197642ef3582734fb0866d65e38cdce02cb331205ac7db78f85ec","observation_id":"f460d0ff-ae55-4d5e-9bba-2f682d51c8b9","resolution":{"observed_at":"2026-08-11T23:47:10.716552Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1712.02170","last_updated":"2017-12-06T13:02:43Z","snapshot_observed_at":"2026-08-15T14:07:23.809064Z","submitted_at":"2017-12-06T13:02:43Z","title":"Detecting Curve Text in the Wild: New Dataset and New Solution","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.02170","snapshot_observed_at":"2026-08-11T23:47:10.286185Z","title":"Detecting curve text in the wild: New dataset and new solu- tion","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.286185Z"},"links":{"cited_paper":"/paper/1712.02170","citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:03190ff9deb0c485c20e6de555a11f1af5e2c34c3f6f98a9e1156a7005e70d7e","observation_id":"c139026f-6b05-4428-96e3-01332c86666b","resolution":{"observed_at":"2026-08-11T23:47:10.286185Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01326","last_updated":"2024-10-11T14:38:40Z","snapshot_observed_at":"2026-08-16T13:46:44.223013Z","submitted_at":"2024-06-03T13:54:05Z","title":"TabPedia: Towards Comprehensive Visual Table Understanding with Concept Synergy","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01326","snapshot_observed_at":"2026-08-11T23:47:10.291965Z","title":"Tabpedia: Towards comprehensive visual ta- ble understanding with concept synergy","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.291965Z"},"links":{"cited_paper":"/paper/2406.01326","citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:b4e2d1a763b6ba53dd5fc950114373b7db92ce5615213c47924f153bb74ace2a","observation_id":"22305401-e979-41f9-b576-29c2070a894e","resolution":{"observed_at":"2026-08-11T23:47:10.291965Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.12628","last_updated":"2024-10-16T14:50:47Z","snapshot_observed_at":"2026-08-18T19:38:48.253429Z","submitted_at":"2024-10-16T14:50:47Z","title":"DocLayout-YOLO: Enhancing Document Layout Analysis through Diverse Synthetic Data and Global-to-Local Adaptive Perception","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.12628","snapshot_observed_at":"2026-08-11T23:47:10.297668Z","title":"Doclayout-yolo: Enhancing document layout analysis through diverse synthetic data and global-to-local adaptive perception","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.297668Z"},"links":{"cited_paper":"/paper/2410.12628","citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:aff765ed2a519d5a333425402858e4eab2ffd22c2d9a3c8b2865fc485e6d8df0","observation_id":"9018851f-08b9-4ffd-ac40-e4d3fc27a634","resolution":{"observed_at":"2026-08-11T23:47:10.297668Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-11T23:47:10.690497Z","title":"3.5 kg”, the true value is “3.5kg","venue":null,"work_id":"aac56d15-42fa-4f2b-b231-d9acb2c90497","year":2020},"citing_paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy","version":3},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-11T23:47:10.303241Z"},"links":{"citing_paper":"/paper/2412.02210"},"observation_digest":"sha256:ffd1116f63a62d0b2875f0a15774b610d6f6985771487f3574488b96b0d6c9bd","observation_id":"9edd0b2c-ef4c-4f1d-8e9b-c4918a5ce15b","resolution":{"observed_at":"2026-08-11T23:47:10.696214Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2412.02210","last_updated":"2024-12-10T05:01:33Z","latest_version":3,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-17T11:15:53.544014Z","submitted_at":"2024-12-03T07:03:25Z","title":"CC-OCR: A Comprehensive and Challenging OCR Benchmark for Evaluating Large Multimodal Models in Literacy"},"reference_resolution":{"displayed":50,"state_counts":{"malformed_identifier":1,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":18,"verified_exact":0,"verified_fuzzy":30},"total_outbound_references":50},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"thesis":"As of 19 August 2026, this Paper Citation Record lists 50 of 50 outbound references and 18 inbound Pith citation observations for arXiv:2412.02210."}