{"as_of":"2026-08-04T09:06:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:5d0643ceb2f1a723d0f94f0c7a4f9e7221bcec872b189d99941e51a67d71225c","coverage":[{"denominator":71,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":71,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-10T16:11:51.138098Z","state":"measured"},{"denominator":73,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":73,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-04T06:34:03.388597+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-02T19:18:53.974983Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2604.12978","snapshot_observed_at":"2026-08-02T19:18:53.974983Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2603.02803","last_updated":"2026-06-16T07:58:47Z","snapshot_observed_at":"2026-08-02T19:18:51.325853Z","submitted_at":"2026-03-03T09:42:43Z","title":"Structure-Aware Text Recognition for Ancient Greek Critical Editions","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-02T19:18:53.974983Z"},"links":{"cited_paper":"/paper/2604.12978","citing_paper":"/paper/2603.02803"},"observation_digest":"sha256:2ce28eb62cc7628f03a747b87e004b43ac3c5d480c06eefd4edfcba28fca7f03","observation_id":"512e2e44-017c-46f2-8a1c-d0bfc246bac5","resolution":{"observed_at":"2026-08-02T19:18:53.974983Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2604.12978","snapshot_observed_at":"2026-08-02T02:58:30.439368Z","title":"GlotOCR Bench: OCR models still struggle beyond a handful of Unicode scripts","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.14041","last_updated":"2026-07-15T17:12:37Z","snapshot_observed_at":"2026-08-02T15:33:36.675572Z","submitted_at":"2026-07-15T17:12:37Z","title":"Multi-Expert Routing for Multi-Domain Low-Resource OCR: A Manchu Case Study","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-02T02:58:30.439368Z"},"links":{"cited_paper":"/paper/2604.12978","citing_paper":"/paper/2607.14041"},"observation_digest":"sha256:ea966fda18db285f956bb87e2ef40ce8bb5539e52dcc28d9317a18b6456210e0","observation_id":"f16b3e8c-c797-4eea-9b51-f0470b427958","resolution":{"observed_at":"2026-08-02T02:58:30.439368Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2604.12978/citation-record","integrity":"/paper/2604.12978/integrity","json":"/paper/2604.12978/citation-record.json","paper":"/paper/2604.12978"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2024.americasnlp-1.10","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"A concise survey of OCR for low-resource lan- guages","venue":null,"work_id":"6bd81c30-c0d3-473c-939d-1f852274d71c","year":2024},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:85b198f4d006483020bff9f98085bca2a89fdb1d4d1030260a8b5444b211ae97","observation_id":"e84247f5-0a98-452f-a6b3-1c29a293ae36","resolution":{"observed_at":"2026-05-10T16:15:34.515602Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Omniglot: Writing systems and languages of the world.https://www.omniglot","venue":null,"work_id":"33561025-bdee-4252-a8db-34e2ef183654","year":2026},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:622e1661347549dc6cde5df24722fc0e470f07e2a1dac8105819984a0ddd99f8","observation_id":"298f45e8-d506-4e19-9349-35212dcce588","resolution":{"observed_at":"2026-05-17T16:31:58.988089Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"CAMIO: A corpus for OCR in multiple languages","venue":null,"work_id":"9560c36c-9de7-4eb8-8370-d4908dd827fa","year":2022},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:126a05c837ed28d47112c7cecba42f015f673ec6ed198ad91fd071b7bebce7a1","observation_id":"6a8a8114-ce3f-4838-a994-af6c1881f5f8","resolution":{"observed_at":"2026-05-17T16:31:58.982445Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2511.21631","last_updated":"2025-11-27T12:16:54Z","snapshot_observed_at":"2026-07-06T22:37:03.716474Z","submitted_at":"2025-11-26T17:59:08Z","title":"Qwen3-VL Technical Report","version":2},"cited_work":{"arxiv_id":"2511.21631","doi":"10.1016/j.neunet.2025.107777","metadata_source":"pith","pith_arxiv_id":"2511.21631","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Qwen3-VL Technical Report","venue":"cs.CV","work_id":"1fe243aa-e3c0-4da6-b391-4cbcfc88d5c0","year":2025},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"cited_paper":"/paper/2511.21631","citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:88d6bb5a19e384905495ceb3b0d739283db06deddc6f6545d3c7c897fa570fc7","observation_id":"4cbcaf47-4feb-47f7-be32-9aac63d3a66d","resolution":{"observed_at":"2026-05-11T09:11:00.926356Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2603.27365","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-02T21:17:24.226483Z","title":"Le Khac, Sanath Narayan, Wamiq Reyaz Para, and Ankit Singh","venue":null,"work_id":"2fe6e984-1a6a-4446-a95f-ab1d0d54844e","year":2026},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:6f2467c85490e7f091c05b33609d4a435f32194a379a3068f797c76758cdfd63","observation_id":"b1926ff8-29f1-4dec-b39b-0b8b0c5bf6e5","resolution":{"observed_at":"2026-05-11T09:11:00.787356Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.06261","last_updated":"2025-12-19T14:25:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-07T17:36:04Z","title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","version":6},"cited_work":{"arxiv_id":"2507.06261","doi":"10.48550/arxiv.2503.19","metadata_source":"pith","pith_arxiv_id":"2507.06261","snapshot_observed_at":"2026-07-11T03:17:51.364436Z","title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","venue":"cs.CL","work_id":"008df105-2fdd-45d8-857a-8e35868aecb6","year":2025},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"cited_paper":"/paper/2507.06261","citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:f3111e99669a40ddab39ee32d55803d4465e74a12ddd7c040b1fd732209cda17","observation_id":"75d5f728-822c-42f4-a8ad-884e074bc2c8","resolution":{"observed_at":"2026-05-11T09:11:00.672945Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.21957","last_updated":"2026-04-03T06:49:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-01-29T16:35:04Z","title":"PaddleOCR-VL-1.5: Towards a Multi-Task 0.9B VLM for Robust In-the-Wild Document Parsing","version":2},"cited_work":{"arxiv_id":"2601.21957","doi":null,"metadata_source":"pith","pith_arxiv_id":"2601.21957","snapshot_observed_at":"2026-07-04T15:09:54.670844Z","title":"PaddleOCR-VL-1.5: Towards a Multi-Task 0.9B VLM for Robust In-the-Wild Document Parsing","venue":"cs.CV","work_id":"aca6cad3-5eb4-48dc-92c1-44a5eb63772e","year":2026},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"cited_paper":"/paper/2601.21957","citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:e8cab82a26c94d08850ddd2d949ae33af6d401d40da0dc6e5e0772c0176a2834","observation_id":"57e8e5cb-549b-4622-b8ca-d743507d41b7","resolution":{"observed_at":"2026-05-11T09:11:00.822354Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Oxford University Press","venue":null,"work_id":"6655d91a-ca57-483b-86ce-0c240d536fba","year":1996},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:84411c1ebc38bc13ddd9df4564452495d820f9b0819313142979fb8c4e151978","observation_id":"4c3c6b31-2930-4d1e-964e-2896b62d372c","resolution":{"observed_at":"2026-05-17T16:31:58.970348Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"sococrbench: An OCR benchmark for social science documents","venue":null,"work_id":"1c9d597a-a7f1-4580-a601-4771f3223cc3","year":2026},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:25a9993699d738f95cb39b0d77bad887c5f94bef53e34f2f79537baab415d787","observation_id":"014bb9bb-913f-42b2-9421-deaf66cf38bc","resolution":{"observed_at":"2026-05-17T16:31:58.990874Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Chandra ocr 2","venue":null,"work_id":"6ff012dc-2aac-4a63-b6b0-687095e4066b","year":2026},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:576fbe0144f82227dd41056bd5090221d1697e65008bf6102a7d7740d60778c5","observation_id":"e5896015-6d87-4bb6-91f1-f75fd6b3ac54","resolution":{"observed_at":"2026-05-17T16:31:58.976048Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.19208","last_updated":"2025-06-24T00:34:55Z","snapshot_observed_at":"2026-07-06T21:46:39.120225Z","submitted_at":"2025-06-24T00:34:55Z","title":"Ancient Script Image Recognition and Processing: A Review","version":1},"cited_work":{"arxiv_id":"2506.19208","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.19208","snapshot_observed_at":"2026-06-30T12:54:40.301459Z","title":"John, and Daqian Shi","venue":null,"work_id":"791bfd85-a8af-4089-b8f3-6ad51e0fb661","year":2025},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"cited_paper":"/paper/2506.19208","citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:ef812784907ae0084882fa1e3e8143b28db31b761173424e50b63aab94daac44","observation_id":"7cdf2315-0e40-4482-8750-20132f81a40e","resolution":{"observed_at":"2026-05-11T09:11:00.762310Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2603.13398","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T10:19:46.900869Z","title":"Qianfan-ocr: A unified end-to-end model for document intelligence","venue":null,"work_id":"daabfed7-5204-4653-861d-2becb02cd711","year":2026},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:a8462f3d8d7332d4d048360c296d0912d3c3874cf31df62c19348eef31ebb6e7","observation_id":"dce7ea07-6e13-4538-863a-12a2e49a882c","resolution":{"observed_at":"2026-05-11T09:11:00.987795Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2603.10910","doi":"10.48550/arxiv.2603.10910.url:http://arxiv","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T17:07:25.785152Z","title":"Glm-ocr technical report.arXiv preprint arXiv:2603.10910","venue":null,"work_id":"c18244fa-5009-4f7a-a303-76b8f016e7a0","year":2026},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:cf26c46b85a27802b57e035d76d4cd9402b5aee5e7f62a3e07d3b7f9d2031ae1","observation_id":"6ab80209-99b6-4621-984a-0beefdd3537b","resolution":{"observed_at":"2026-05-11T09:11:01.010104Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"HarfBuzz: A text shaping engine","venue":null,"work_id":"4924324d-b116-4102-aa77-6c5702758541","year":2026},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:32731634063af9e672a0321d47847ae3c6fad06a6f32c30151cc45b5907b8bf7","observation_id":"4a5b9367-0975-418b-97ca-729603501d4f","resolution":{"observed_at":"2026-05-17T16:31:58.979015Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"OCRBench v2: An improved benchmark for evaluating large multimodal models on visual text localization and reasoning","venue":null,"work_id":"a64e9991-1e76-465b-b017-e3cd0b45ce08","year":2025},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:980c3029e235ca35b5a223a5847fb628d8f745500046e9b4755afd6d6b46f088","observation_id":"9dedfff2-0801-42cb-af47-01afb3b1feb8","resolution":{"observed_at":"2026-05-17T16:31:59.014183Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gemini 3.1 flash-lite: Built for intelligence at scale","venue":null,"work_id":"5d1d2016-6cec-48be-a77e-edfe4da39f5d","year":2026},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:127714ca7952f4c33d837f64a97c37a688f59bd94ef47eb3c05609a65cb0047c","observation_id":"45ac6fb2-31df-4aa8-8a15-edc6e6d93a98","resolution":{"observed_at":"2026-05-17T16:31:59.038166Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Google fonts","venue":null,"work_id":"7b881860-4f8f-477f-8263-328d870e0c44","year":2026},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:62583a1071e5184afb732d1b9368bb7d11960fc0e7f258f5906c88de67fb6876","observation_id":"7a41ce9f-d479-4186-9a86-e3e56be7e695","resolution":{"observed_at":"2026-05-17T16:31:59.041275Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.00414","last_updated":"2025-04-01T04:21:34Z","snapshot_observed_at":"2026-07-06T21:02:08.295333Z","submitted_at":"2025-04-01T04:21:34Z","title":"Multimodal LLMs for OCR, OCR Post-Correction, and Named Entity Recognition in Historical Documents","version":1},"cited_work":{"arxiv_id":"2504.00414","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2504.00414","snapshot_observed_at":"2026-07-02T02:26:26.386515Z","title":"Kaiming He, Xiangyu Zhang, Shaoqing Ren, and Jian Sun","venue":null,"work_id":"45834e3f-0fc0-40ca-a356-d48f8ec1143f","year":2025},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"cited_paper":"/paper/2504.00414","citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:b6cee2c6cb4975cbc0ae8f010c75cd1c0d9618e9dd14ef889d796c9338baffa1","observation_id":"005ae4b2-9b68-435c-a340-9665a9ec49f7","resolution":{"observed_at":"2026-05-11T09:11:00.976237Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2208.14558","last_updated":"2023-03-24T21:49:21Z","snapshot_observed_at":"2026-07-06T13:47:05.262612Z","submitted_at":"2022-08-30T22:36:19Z","title":"Augraphy: A Data Augmentation Library for Document Images","version":2},"cited_work":{"arxiv_id":"2208.14558","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2208.14558","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Augraphy: A data augmentation library for document images","venue":null,"work_id":"027c5d5d-6542-427f-a509-08d6ac2be2a9","year":2023},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"cited_paper":"/paper/2208.14558","citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:a1b4a5ffdd352bf9047c0dd50ba3627f1042ab1fa085639db1ee535dff8b12c0","observation_id":"59d43229-0b15-4faf-ac88-ae9a1573e9dc","resolution":{"observed_at":"2026-05-11T09:11:00.960816Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Synthetic data for text localisation in natural images","venue":null,"work_id":"369db87a-7111-4724-8c66-3cbb03aebe92","year":2016},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:ad6702c14b61b3f020a8b7528dd7c6878e117d0b9bf09704d7b899d6a8715901","observation_id":"2fabebb9-010f-4c75-b36f-9e4023874cee","resolution":{"observed_at":"2026-05-17T16:31:59.044656Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.12766","last_updated":"2025-05-19T06:45:18Z","snapshot_observed_at":"2026-07-06T21:26:01.534136Z","submitted_at":"2025-05-19T06:45:18Z","title":"Reasoning-OCR: Can Large Multimodal Models Solve Complex Logical Reasoning Problems from OCR Cues?","version":1},"cited_work":{"arxiv_id":"2505.12766","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.12766","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Reasoning-OCR: Can large multimodal models solve complex logical reasoning problems from ocr cues?","venue":null,"work_id":"1f56692b-f9f4-4a5a-b519-0cd19e7a4e3b","year":2025},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"cited_paper":"/paper/2505.12766","citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:566f230814297c5db407b733cd0cc5f940df8ff22693b1300c18d91de44b2440","observation_id":"5dade1ca-1000-4b86-8708-6d2536cc70fc","resolution":{"observed_at":"2026-05-11T09:11:00.667074Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2025.findings-acl.1135","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"KITAB-bench: A comprehensive multi-domain benchmark for Arabic OCR and document understanding","venue":null,"work_id":"dd8cdc9e-6af6-4d73-931e-6158b17521c2","year":2025},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:ced46770957dbb1dba0ffda78a02d7afde1f001574666a9e90b6cb7647f8032b","observation_id":"3b632bdd-f57f-4c86-9c1f-ac37574b926c","resolution":{"observed_at":"2026-05-10T16:15:34.513700Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2511.19575","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T17:07:25.779583Z","title":"Hunyuanocr technical report","venue":null,"work_id":"2380b8a5-8713-4b4a-8922-7b1baf35db89","year":2025},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:d6416ae6bc37fe515e73d343c73026056e1a52c8be32fa140df0948f174cd364","observation_id":"6d8cf427-fdca-4d8e-b481-57fe8af3c72e","resolution":{"observed_at":"2026-05-11T09:11:00.684279Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Generating errors: OCR post-processing for Icelandic","venue":null,"work_id":"90066417-faa4-4aac-862a-46c275e95b33","year":2023},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:46e8967e4becfcecc6a10bd73724fc6c64841daebf506be5b5458c3f3b773b3d","observation_id":"83dc8fc1-1c03-4a67-8b68-c124bf1d6e82","resolution":{"observed_at":"2026-05-17T16:31:59.047673Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2025.acl-long.1260","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Evaluating multimodal language models as visual assistants for visually impaired users","venue":null,"work_id":"8b49b1f9-6c5a-4530-9c6b-935b161555db","year":2025},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:0915d42650f07fd03d78555c193d16f2e1d781e8a930df50a864c006dad0f7cf","observation_id":"2a915302-97b7-409d-add3-d6160151b20e","resolution":{"observed_at":"2026-05-10T16:15:34.517586Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2023.findings-emnlp.410","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T01:41:29.529504Z","title":"G lot LID : Language identification for low-resource languages","venue":null,"work_id":"1d380262-9360-4840-bcdc-219e0753cb24","year":2023},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:6b2ed270f2614f96df71512154c008ee43d14afa0b67beb9100f623038463726","observation_id":"301576a3-3422-4405-bbe8-6462be418191","resolution":{"observed_at":"2026-05-10T16:15:34.528639Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"GlotCC: An open broad- coverage commoncrawl corpus and pipeline for minority languages","venue":null,"work_id":"22f85d2b-9c16-4472-ae94-4bc5f9253e9b","year":2024},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:a20868ad28686d1f125d9d3471b8b862dec9018c545376eb98207843f7179ffd","observation_id":"ecba8fb4-a329-41f1-9411-103c06a25ab9","resolution":{"observed_at":"2026-05-17T16:31:59.050957Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"GlotScript: A resource and tool for low resource writing system identification","venue":null,"work_id":"03683171-e62c-4f46-baf3-83f6f10e6d02","year":2024},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:eb2f41dcd0bde5dbf32e76a2ef2a9957f847161f9b607d1a9ea2f9d926e1fdba","observation_id":"2380fa77-bada-438c-8b20-1ec5c0ec6989","resolution":{"observed_at":"2026-05-17T16:31:59.057150Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Nayana OCR: A scalable framework for document OCR in low-resource languages","venue":null,"work_id":"3ef48daa-52d0-463c-90e6-25e824a2f34a","year":2025},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:ede4822a4f5ac70334777733f4458da8f2fe637b701c6103ff84b74c95ef4dc9","observation_id":"ed5d4852-67e4-4643-af65-974f4fc3346c","resolution":{"observed_at":"2026-05-17T16:31:59.020831Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2025.lm4uc-1.11","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"URL https://aclanthology.org/2025.lm4uc-1","venue":null,"work_id":"b7e2d684-64dc-4be0-b5d6-113945633f1f","year":2025},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:e9e7a137e6d1315e18b1a8bad09609a086a6afebdce221662505cce9aeb549bf","observation_id":"188e9b80-d3f3-4e3b-a4f5-513514f0967f","resolution":{"observed_at":"2026-05-10T16:15:34.530337Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"FinePDFs","venue":null,"work_id":"e9df33bd-e3ff-4cab-8138-fce95c0b867c","year":2025},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:16d3bd1818d10e80d5b0197f927791de90f209ac8cc42632ed454c5c8198e2ac","observation_id":"f45aa312-864e-4786-9be0-0b13f98843f2","resolution":{"observed_at":"2026-05-17T16:31:59.031465Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2512.02498","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T09:59:45.152724Z","title":"InICASSP 2023-2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pages 1–5","venue":null,"work_id":"cb5539a0-6843-4702-bbf3-dfe7b61e5cbb","year":2025},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:d01b68ff23bf8c498cdbf92b6d9048e0d92305d083339de5f3ddc903e9811dc9","observation_id":"6b046eb7-d195-4af4-9f6b-81e160d36217","resolution":{"observed_at":"2026-05-11T09:11:00.706353Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2506.05218","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T10:19:46.911382Z","title":"arXiv preprint arXiv:2506.05218 , year=","venue":null,"work_id":"e1783180-c7e4-4eda-92a0-bea7df4e8d95","year":2025},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:1a8d305547789cb58b8bb30b23154e946fdb600998562413d0f96897a24c8d06","observation_id":"fec9378f-f9f3-4283-9e59-c056859f45d1","resolution":{"observed_at":"2026-05-11T09:11:00.892273Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2602.21042","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Omniocr: Generalist ocr for ethnic minority languages","venue":null,"work_id":"a8bd5a47-907a-476c-adab-4a0f1af0293c","year":2026},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:89a5c044549d7e83dc8f5aea3b047007c1941d0919276e546e6d5294f2bfa4a3","observation_id":"99b5f380-cdbe-42b6-85d2-519b20a84b58","resolution":{"observed_at":"2026-05-11T09:11:00.946595Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1038/s41597-024-03918-5","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Ancient yi script handwriting sample repository.Scientific Data, 11(1):1183","venue":null,"work_id":"20440cd5-36d5-4311-96a7-5bf165e7d2d6","year":2024},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:ac672bfd3f14d1909f1b1c9d184da5721177dd6a34090d350042b3cdd79f1ca9","observation_id":"94aa3bc1-ae68-440a-9008-24477f1c8081","resolution":{"observed_at":"2026-05-10T16:15:34.526872Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1007/s11432-024-4235-6","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-01T05:45:25.355196Z","title":"Ocrbench: on the hidden mystery of ocr in large multimodal models","venue":null,"work_id":"b744bc1a-628b-43e1-8704-7addfd21cfcf","year":1919},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:d968ae01dc4f52bd0693f47be1692e3760188f0055b1f4cfc219cadd3a334f8e","observation_id":"8747d207-cf21-4d32-9445-9b388f070804","resolution":{"observed_at":"2026-05-10T16:15:34.519686Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Multiple attentional aggregation network for handwritten Dongba character recognition.Expert Systems with Applications, 213:118865","venue":null,"work_id":"752f220b-a1b3-441f-8aae-98d45f7e3e7f","year":null},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:1e62a9f59b4ac904ff1c7c9214eae7f5b6a9d75aaff2690a50e09eec07e31121","observation_id":"eda1b8f9-9dc2-4ce1-beed-561d6c3a4dd8","resolution":{"observed_at":"2026-05-17T16:31:59.054132Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2022.118865","doi":"10.1016/j.eswa.2022.118865","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"doi: https://doi.org/10.1016/j.eswa.2022.118865","venue":null,"work_id":"2b7bf312-06f6-4d96-84cc-5ce89a1876b4","year":2022},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:adc5eb624baebbe037a19120cac52a2128d134fae2afbe8f174aab9ba3778a7e","observation_id":"09580ad4-a8e6-471f-98bc-ad5809d5eede","resolution":{"observed_at":"2026-05-10T16:15:34.522467Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2601.16113","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T09:59:45.223867Z","title":"synthocr-gen: A synthetic OCR dataset generator for low-resource languages- breaking the data barrier","venue":null,"work_id":"ab4ce312-8b1d-40ed-b81b-754660d662ac","year":2026},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:a7e17d6eaeaa0df3c1c418dead2be6120607f02b9e8604c9693e8b6969692f2c","observation_id":"dff3b822-5cbe-4b10-b537-6823185f1838","resolution":{"observed_at":"2026-05-11T09:11:00.905155Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Nanonets-OCR2: A model for transforming documents into structured markdown with intel- ligent content recognition and semantic tagging","venue":null,"work_id":"cd79ab69-f7d2-49e5-a30c-e9c90258d394","year":2025},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:4127e5195f4e519215c85da452fc3b27aef1a7ba282c57d966f41fc10644ac20","observation_id":"4a20ab4f-45f8-4e67-b12b-2e194a37c1ba","resolution":{"observed_at":"2026-05-17T16:31:59.019229Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Nemotron ocr v2","venue":null,"work_id":"33d8f895-6a85-4c80-9c69-de33a663a4de","year":2026},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:8077ec9cce61b2855863f4b6edbf768c388db9a96d0f82cbb28e01c5e7f2ecb8","observation_id":"5e363f1d-1af5-4cef-9135-0dd5fd6d38d1","resolution":{"observed_at":"2026-05-17T16:31:59.010163Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":"2303.08774","doi":"10.1002/tea.20265","metadata_source":"pith","pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"GPT-4 Technical Report","venue":"cs.CL","work_id":"b928e041-6991-4c08-8c81-0359e4097c7b","year":2023},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:ad4e14eff40101c38a7590d5aa31357ec2ecc11ea902af73de13b356fc9880b3","observation_id":"1659255d-70fa-40f4-9c46-258d0c221f48","resolution":{"observed_at":"2026-05-11T09:11:00.861358Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"OmniDocBench: Benchmarking diverse pdf document parsing with comprehensive annotations","venue":null,"work_id":"ad304760-9ce3-487c-b270-6f0a48b8173e","year":2025},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:35817c533b1a23869ba58da81b358b3af4f2fe96fff01d3dbd120551dc1a20fc","observation_id":"f584528a-ec65-4f90-a583-af0cc37ed363","resolution":{"observed_at":"2026-05-17T16:31:58.991122Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Fineweb2: One pipeline to scale them all — adapting pre-training data processing to every language","venue":null,"work_id":"7eb65adc-ce8f-4fbb-bfb1-a51111acbb47","year":2025},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:13398bc0689f8e02840c2fafbf8823a2d0e1fa280388ea40ef60c3a9aea861c5","observation_id":"e1a3a301-cc30-4a9d-86c5-15ddc991e1a2","resolution":{"observed_at":"2026-05-17T16:31:58.998741Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2502.18443","doi":"10.48550/arxiv.2502.18443","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T17:07:25.776831Z","title":"olmocr: Unlocking trillions of tokens in pdfs with vi- sion language models.arXiv preprint arXiv:2502.18443, 2025a","venue":null,"work_id":"bb9c92f3-5210-40cf-b3d5-7463913e3248","year":2025},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:a8ca5ea524f67ca8733c1bfcd381843b87b725f9833044d0f297246c6aa414dc","observation_id":"15ad61de-6bbc-4489-9f75-febf893ebb10","resolution":{"observed_at":"2026-05-11T09:11:00.749157Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"olmOCR 2: Unit test rewards for document ocr","venue":null,"work_id":"7dfacf14-44a9-4563-b710-5f88220960d0","year":null},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:25c79d78bff7b548578784688edcb0e7935e73425521e400da422ce4cbe3f2c5","observation_id":"9458ad08-840c-4125-b067-ed4c2ab968e6","resolution":{"observed_at":"2026-05-17T16:31:58.956324Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.19817","doi":"10.48550/arxiv.2510.19817","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T17:07:25.791394Z","title":"olmOCR 2: Unit test rewards for document OCR","venue":null,"work_id":"7a5b54b8-26b0-4fe0-ba2f-39b063ae3aa7","year":2025},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:d2e52c9559e64b821e8e893f47b08f4bfce011b91e36c2514c8a03a28b52741d","observation_id":"adbb9938-0f3b-4a9f-8d02-21950c0f475a","resolution":{"observed_at":"2026-05-11T09:11:01.000428Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Aksharamukha: Script conversion web tool","venue":null,"work_id":"5751023b-20ab-4256-ac87-14667b7ba391","year":2024},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:8c9a49d13c8370cd9cb12b7557c6201bfdbd03af6ae5fc761f451543eae46851","observation_id":"2d7d9a2e-ea3e-4e67-847c-2f284aff824a","resolution":{"observed_at":"2026-05-17T16:31:58.979328Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Rolmocr: A faster, lighter open-source ocr model","venue":null,"work_id":"2dc7f0f3-0e81-4944-9e41-08649564c302","year":2025},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:13c948b84b69d0644243eddbb41fbee908b299e03b4fa1c0979cd5d1d4a9439d","observation_id":"34220b88-eb5c-4731-9e44-a43cd7d89cbf","resolution":{"observed_at":"2026-05-17T16:31:58.962325Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2205.02543","last_updated":"2022-05-05T10:07:57Z","snapshot_observed_at":"2026-07-06T13:06:54.455644Z","submitted_at":"2022-05-05T10:07:57Z","title":"OCR Synthetic Benchmark Dataset for Indic Languages","version":1},"cited_work":{"arxiv_id":"2205.02543","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2205.02543","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Ocr synthetic benchmark dataset for indic languages","venue":null,"work_id":"553d25bd-9fca-4199-b454-7ad197f324c5","year":2022},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"cited_paper":"/paper/2205.02543","citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:37ac63f09f2635659bf719a4e5a3f5da82bc3276b4231a46f641fa964d2ef1d8","observation_id":"f2fbdcff-5de9-48dd-b5cd-783445811c0e","resolution":{"observed_at":"2026-05-11T09:11:00.659346Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Printed ocr for extremely low-resource indic languages","venue":null,"work_id":"64220f30-e8e1-4f9e-9767-1f75f36dfa32","year":2024},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:932c0cff82b0edbe3c18308a08db527994b42a932f0b1438daf62f3edad60a26","observation_id":"ab522cd9-2039-4652-99e1-73e49986497c","resolution":{"observed_at":"2026-05-17T16:31:59.004988Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"4904.379288","doi":"10.1145/3774904.3792887","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"GlotWeb: Web indexing for minority languages","venue":null,"work_id":"cd6a5a41-d596-498c-9d2f-3449eee48fd9","year":2026},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:25da5f1834146620e13d4d4951bb1a6bf260a942ca4d77bc6568c4d5eb870882","observation_id":"0ad17c4e-e57a-4918-ba84-89776df6a747","resolution":{"observed_at":"2026-05-10T16:15:34.524951Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Deciphering the underserved: Benchmarking llm ocr for low-resource scripts","venue":null,"work_id":"6995aaea-9c40-4476-8f53-511653102cfa","year":2024},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:d32f16dd91e4a73ae4cfec2da2262b83783c7f7be5cb77f745b59fa119801cf2","observation_id":"a0d079d1-fd87-4de2-a2ad-9a418e762214","resolution":{"observed_at":"2026-05-17T16:31:58.963293Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.14251","last_updated":"2026-06-30T08:51:50Z","snapshot_observed_at":"2026-08-03T11:27:18.044435Z","submitted_at":"2026-01-20T18:58:32Z","title":"LightOnOCR: A 1B End-to-End Multilingual Vision-Language Model for State-of-the-Art OCR","version":2},"cited_work":{"arxiv_id":"2601.14251","doi":"10.48550/arxiv.2601.14251","metadata_source":"pith","pith_arxiv_id":"2601.14251","snapshot_observed_at":"2026-07-10T17:07:25.753449Z","title":"Lightonocr: A 1b end-to-end multilingual vision-language model for state-of-the-art ocr.arXiv preprint arXiv:2601.14251","venue":"cs.CV","work_id":"5fe5d497-a726-4f15-920b-e4da9a150630","year":2026},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"cited_paper":"/paper/2601.14251","citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:6d8e35f1ea71e446659b15dd371fac6004ae02edfe5460f4e1d783d6ab1b9b7e","observation_id":"413b88c0-e8ab-45bf-8c78-9088ff1646a8","resolution":{"observed_at":"2026-07-01T02:17:18.970700Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"FreeType: A free, high-quality and portable font engine","venue":null,"work_id":"994120b9-553c-4d73-829b-ed501bc707c9","year":2024},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:46869ce7d1af2918f4dbcecba7c3e58a72b6910aa0268a038e0ec95abd1deb64","observation_id":"602afba4-f36a-4e16-8052-d502d90416ab","resolution":{"observed_at":"2026-05-17T16:31:58.999614Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Ocr uv scripts","venue":null,"work_id":"b84c182c-156e-4c96-9db9-300b031c3144","year":2026},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:d4030723b6dc4fee57bd120188ea102236efdc1d3a065c678a60a14b9eeea02b","observation_id":"326ad61d-948e-4cb6-b1c8-3af4d1ee85c4","resolution":{"observed_at":"2026-05-17T16:31:58.985325Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2601.20552","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T17:07:25.727485Z","title":"Deepseek-ocr 2: Visual causal flow","venue":null,"work_id":"9886dce9-776e-4ad6-b96c-fbf9037ca41c","year":2026},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:baf8204a66a26945899099c7c3660364d683f1c0fbc1a20ee0f2d456535051c8","observation_id":"17cd64eb-2c40-4a5d-87cf-97129f03b0f0","resolution":{"observed_at":"2026-05-11T09:11:00.850411Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Wikisource: The free online library","venue":null,"work_id":"60ebb7f3-dffa-4989-9d3d-a323452c64bf","year":2026},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:429ff59f03d1be036dccb9fc11bcf6e9edced23ec1a0afac3367646cde163d23","observation_id":"bf3d877c-63ff-46ec-858a-8aabe5bd5892","resolution":{"observed_at":"2026-05-17T16:31:59.063232Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Wiktionary, the free dictionary","venue":null,"work_id":"9a425e53-9eae-4cc4-a7a1-9b12a96de7e8","year":2026},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:7671045f97243ee96728c2c04d6f8ff76341a0046fccbe0456dc4ff44d9321c0","observation_id":"db79f326-6ddf-4eab-8a3e-a1e17a69ff98","resolution":{"observed_at":"2026-05-17T16:31:59.060242Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2603.01840","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T10:19:46.941324Z","title":"Firered-ocr technical report","venue":null,"work_id":"36c46655-2f26-436d-a19c-295616f0ae96","year":2026},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:d8ddec2bd01b4c8b68d720b9510c080b3cb180b346678753024f96c4665913e5","observation_id":"cfedb454-5d1a-4687-8074-2841fc172d49","resolution":{"observed_at":"2026-05-11T09:11:00.729347Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"CC-OCR: A comprehensive and challenging ocr benchmark for evaluating large multimodal models in literacy","venue":null,"work_id":"aad34d02-64d5-4059-9812-9d69c81bbb62","year":2025},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:a8385a6c668e3fb52931e2e56148eda1a08181eab78d84804f0e3c62a3a67bf9","observation_id":"ddcb5625-56f2-41a9-b05c-525d262eb11e","resolution":{"observed_at":"2026-05-17T16:31:58.953141Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2602.03693","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Ocrturk: A comprehensive ocr benchmark for turkish","venue":null,"work_id":"d7c4d4db-45c4-4841-aa66-648eed9ecf2b","year":2026},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:c7635cf3985c139bfe3f1fcb9b8a096875fcfbbb9cd31b4e5eeac42eefdb2991","observation_id":"2c64ecf6-ab8a-407c-97d4-f3aae54f9db0","resolution":{"observed_at":"2026-05-11T09:11:00.937470Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Synthtiger: Synthetic text image generator towards better text recognition models","venue":null,"work_id":"c3e0a153-04a3-4d7a-ae0e-9cf88a0fcc50","year":2021},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:1a09d674bc4885f51bede27c4c397c2048237a456dea1482aafde3a852f2027b","observation_id":"fc332cbe-aaf2-45d2-8451-95008650bf64","resolution":{"observed_at":"2026-05-17T16:31:58.964928Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Tibetanmnist: Tibetan handwritten digit dataset","venue":null,"work_id":"75d5129a-2d70-496c-967a-788ce5da8fda","year":null},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:29edca524262d4b1c884fb904a8cd8de084d39652a80aa784101c206de6c095c","observation_id":"22852c4f-6e5a-4550-abc2-b72918e8fce7","resolution":{"observed_at":"2026-05-17T16:31:58.985840Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"292ab763-5fa0-4013-8bab-f2bb24d99d16","year":null},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:db5f1ab58f507c4474cbd73235c9085a575f73d2d0c7b5b29d8862aaf3c8761d","observation_id":"9ddf5345-d366-41fb-ba48-8cd391494a18","resolution":{"observed_at":"2026-05-17T16:31:59.002532Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2603.13032","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-01T21:16:14.035094Z","title":"Multimodal ocr: Parse anything from documents","venue":null,"work_id":"83409bb9-0ee5-4da9-a750-da6424ff01f5","year":2026},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:ae11ed330bd5c607a838873fbed0ecff921bc33c59ba663a789244b6c98a5eb7","observation_id":"571a22d3-8b77-4e3a-9e98-d19aad372a29","resolution":{"observed_at":"2026-05-11T09:11:00.878157Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"2edc0939-37d6-493c-8e02-22679cfe4683","year":null},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:7b17ab9c923aac8425b2b3615277299ae957d902b1c91b3be0ff29cbdce1004b","observation_id":"e819ab46-4ce1-443d-ae91-be15f08b3cb9","resolution":{"observed_at":"2026-05-17T16:31:59.017424Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"add69098-4376-442d-93a3-10d28b3fa1cf","year":null},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:48c861a51284e0b8dc019c558cb3e2640781ec1f848668f5cdceb68b60d6c0e7","observation_id":"558033b1-05dd-4c16-9e7d-695197a4582f","resolution":{"observed_at":"2026-05-17T16:31:59.023829Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"50df2abb-1115-42c6-a5c9-7b146bcc948e","year":null},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:9db2b06790d97032869cadaaec0ce9d6c7bba23a533fc3b1051ff6da53c37783","observation_id":"3a56d78b-9768-4c15-90e9-b80bf337c16b","resolution":{"observed_at":"2026-05-17T16:31:59.034716Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"523dbf46-5b35-421b-aa19-7ead9f6b1c58","year":null},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:4f42f00b61ea1e94b7512d6fb6a4cae1767d84557f3b139b98f6d0d432116994","observation_id":"2f6d2071-15bb-4f76-8ad3-2c1a7226a654","resolution":{"observed_at":"2026-05-17T16:31:59.028054Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Additionally, at the glyph level during rendering, character spacing is perturbed by −2 to +4 pixels, each glyph is independently dilated (prob","venue":null,"work_id":"ab8e44fb-5521-41c5-b03b-6e9a968650e0","year":null},"citing_paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-05-10T16:11:51.138098Z"},"links":{"citing_paper":"/paper/2604.12978"},"observation_digest":"sha256:a73b9ff3037575d10f4cff66d6f3061938ec8983d6ef0a0cfa5b1e20dbc66c05","observation_id":"703fd8e8-9b11-49d6-bee3-f117f7969b6b","resolution":{"observed_at":"2026-05-17T16:31:59.006095Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2604.12978","last_updated":"2026-04-14T17:12:41Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-07-06T23:01:05.634599Z","submitted_at":"2026-04-14T17:12:41Z","title":"GlotOCR Bench: OCR Models Still Struggle Beyond a Handful of Unicode Scripts"},"reference_resolution":{"displayed":71,"state_counts":{"malformed_identifier":1,"metadata_mismatch":3,"parse_uncertain":0,"unresolved":5,"verified_exact":30,"verified_fuzzy":32},"total_outbound_references":71},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"thesis":"As of 4 August 2026, this Paper Citation Record lists 71 of 71 outbound references and 2 inbound Pith citation observations for arXiv:2604.12978."}