{"as_of":"2026-08-07T04:16:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:40067d51626b1ca9cb525176d25479e3c4e83f89c578a7d4d1aa5054674c15bc","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":30,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":30,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":30,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":30,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T04:08:48.010046Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-23T19:43:23.773160Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-04T10:28:00.554323Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":"2403.04473","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":"612fe156-781e-49f3-bac5-03fcb13d115f","year":2024},"citing_paper":{"arxiv_id":"2306.13549","last_updated":"2024-11-29T15:51:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-23T15:21:52Z","title":"A Survey on Multimodal Large Language Models","version":4},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-16T02:56:41.658658Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2306.13549"},"observation_digest":"sha256:2b0756ca9cf99f6f0142b7a47494267041f8526bd046ac7531939c00b2267952","observation_id":"541007d1-b508-4c6d-8895-89630c02dfa3","resolution":{"observed_at":"2026-05-16T02:56:42.414279Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-04T10:28:00.554323Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":"2403.04473","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":"612fe156-781e-49f3-bac5-03fcb13d115f","year":2024},"citing_paper":{"arxiv_id":"2404.16821","last_updated":"2024-04-29T20:24:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-04-25T17:59:19Z","title":"How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites","version":2},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-05-12T20:58:58.849040Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2404.16821"},"observation_digest":"sha256:685fabdcb421818dbe14f38898568b3516b0dd59b7fa01cd719c120988702772","observation_id":"dce1f58e-15c5-4d58-9261-e90500800966","resolution":{"observed_at":"2026-05-12T20:58:59.111646Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-04T10:28:00.554323Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":"2403.04473","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":"612fe156-781e-49f3-bac5-03fcb13d115f","year":2024},"citing_paper":{"arxiv_id":"2407.03320","last_updated":"2024-07-03T17:59:21Z","snapshot_observed_at":"2026-08-04T22:09:42.241578Z","submitted_at":"2024-07-03T17:59:21Z","title":"InternLM-XComposer-2.5: A Versatile Large Vision Language Model Supporting Long-Contextual Input and Output","version":1},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-05-17T10:46:28.447347Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2407.03320"},"observation_digest":"sha256:9010a7aee84c0ad9ccff750aa2af1843bbb408075ea759f6634038fcbaf0c51d","observation_id":"77c8df6e-1d1c-461e-89b0-5599283f556b","resolution":{"observed_at":"2026-05-17T10:46:28.774762Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-04T10:28:00.554323Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":"2403.04473","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":"612fe156-781e-49f3-bac5-03fcb13d115f","year":2024},"citing_paper":{"arxiv_id":"2408.01800","last_updated":"2024-08-03T15:02:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-03T15:02:21Z","title":"MiniCPM-V: A GPT-4V Level MLLM on Your Phone","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-05-10T21:07:31.387726Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2408.01800"},"observation_digest":"sha256:eec580dfce9a90ba63ce823355ec8ab45fa175168279d3bd3794d5d760ea5397","observation_id":"d0d22a99-2dca-4228-b9e2-14d19b081665","resolution":{"observed_at":"2026-05-10T21:07:32.096671Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-04T10:28:00.554323Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":"2403.04473","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":"612fe156-781e-49f3-bac5-03fcb13d115f","year":2024},"citing_paper":{"arxiv_id":"2408.13257","last_updated":"2025-02-05T08:44:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-23T17:59:51Z","title":"MME-RealWorld: Could Your Multimodal LLM Challenge High-Resolution Real-World Scenarios that are Difficult for Humans?","version":3},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-16T07:59:32.638758Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2408.13257"},"observation_digest":"sha256:3f8e517c1effd13bc929d545fe7b38029e505054c1130cfd49fcef7b733640fc","observation_id":"9f91f9c5-3858-4714-b421-252add1f24f6","resolution":{"observed_at":"2026-05-16T07:59:32.827628Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-04T10:28:00.554323Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":"2403.04473","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":"612fe156-781e-49f3-bac5-03fcb13d115f","year":2024},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-01T15:04:20.648996Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:d95f244f69d0aa63094d223cbf45b5ad29fa4cbf461d032bc645ad22a23097a0","observation_id":"3032a2af-cde3-4694-8376-af8e8cfababc","resolution":{"observed_at":"2026-05-17T20:50:57.878737Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-04T10:28:00.554323Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":"2403.04473","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":"612fe156-781e-49f3-bac5-03fcb13d115f","year":2024},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:eeea753060fdba80e5c63639c269c75d3f6e67dbfd5e5d4cbd949f2b86d99260","observation_id":"8a11461b-8a19-4f64-b893-9cb44861c518","resolution":{"observed_at":"2026-05-23T19:43:23.776809Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-04T10:28:00.554323Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":"2403.04473","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":"612fe156-781e-49f3-bac5-03fcb13d115f","year":2024},"citing_paper":{"arxiv_id":"2410.10594","last_updated":"2025-03-02T01:19:51Z","snapshot_observed_at":"2026-08-02T13:24:00.339129Z","submitted_at":"2024-10-14T15:04:18Z","title":"VisRAG: Vision-based Retrieval-augmented Generation on Multi-modality Documents","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-16T15:37:25.781240Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2410.10594"},"observation_digest":"sha256:74c8da7e1536d2880658395378800ef11054eaa8477dccadc3ffa09aa03cee5b","observation_id":"fe01c260-d247-4af5-b785-ac057b13bc24","resolution":{"observed_at":"2026-05-16T15:37:25.911108Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-04T10:28:00.554323Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":"2403.04473","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":"612fe156-781e-49f3-bac5-03fcb13d115f","year":2024},"citing_paper":{"arxiv_id":"2410.21169","last_updated":"2026-04-04T17:04:02Z","snapshot_observed_at":"2026-07-06T19:40:52.844113Z","submitted_at":"2024-10-28T16:11:35Z","title":"Document Parsing Unveiled: Techniques, Challenges, and Prospects for Structured Information Extraction","version":5},"reference_index":142,"source":"pdf_text","source_observed_at":"2026-05-23T19:15:21.695801Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2410.21169"},"observation_digest":"sha256:3718ae02402b2ef9018a5c4f3d7269c7cce7a62886621dfb2c1b433b27ddeae9","observation_id":"ca16e56a-e147-4132-b09d-d7c651eb04ce","resolution":{"observed_at":"2026-05-23T19:15:47.281080Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-04T10:28:00.554323Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":"2403.04473","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":"612fe156-781e-49f3-bac5-03fcb13d115f","year":2024},"citing_paper":{"arxiv_id":"2501.00321","last_updated":"2025-06-05T02:59:05Z","snapshot_observed_at":"2026-08-02T18:54:27.250149Z","submitted_at":"2024-12-31T07:32:35Z","title":"OCRBench v2: An Improved Benchmark for Evaluating Large Multimodal Models on Visual Text Localization and Reasoning","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-17T20:33:26.613927Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2501.00321"},"observation_digest":"sha256:969988898eea506989c75a391dafdaa32f5bbdc69e010c87280a8b8483c995a9","observation_id":"7731a63b-db2d-49a7-ae06-be6343d86ea3","resolution":{"observed_at":"2026-05-17T20:33:26.724787Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-04T10:28:00.554323Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":"2403.04473","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":"612fe156-781e-49f3-bac5-03fcb13d115f","year":2024},"citing_paper":{"arxiv_id":"2503.12937","last_updated":"2025-08-04T04:22:09Z","snapshot_observed_at":"2026-08-06T21:34:54.534751Z","submitted_at":"2025-03-17T08:51:44Z","title":"R1-VL: Learning to Reason with Multimodal Large Language Models via Step-wise Group Relative Policy Optimization","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-16T15:04:22.690503Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2503.12937"},"observation_digest":"sha256:22ab3d1d5a4561f1d6c9c327cbca2ce4bc739947b5834602030beca9f4fd7d97","observation_id":"331caa8d-4bf1-46d3-9584-531062fe2fce","resolution":{"observed_at":"2026-05-16T15:04:22.851273Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-04T10:28:00.554323Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-08-07T04:08:48.010046Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.11515","last_updated":"2025-06-13T07:16:41Z","snapshot_observed_at":"2026-08-07T04:02:11.274253Z","submitted_at":"2025-06-13T07:16:41Z","title":"Manager: Aggregating Insights from Unimodal Experts in Two-Tower VLMs and MLLMs","version":1},"reference_index":98,"source":"pdf_text","source_observed_at":"2026-08-07T04:08:48.010046Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2506.11515"},"observation_digest":"sha256:b98d11d8e0c8ba9b4f9a4ba1f375e083b116524154d4b4f1e810dc14b495e8cc","observation_id":"55ef876c-a123-4673-beb6-714e832d43db","resolution":{"observed_at":"2026-08-07T04:08:48.010046Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-04T10:28:00.554323Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-08-06T23:47:38.535627Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.21600","last_updated":"2025-06-19T07:16:18Z","snapshot_observed_at":"2026-08-06T23:42:13.492339Z","submitted_at":"2025-06-19T07:16:18Z","title":"Structured Attention Matters to Multimodal LLMs in Document Understanding","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-06T23:47:38.535627Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2506.21600"},"observation_digest":"sha256:4a2de4e63d086d40b758c8bf308fea398d06d019996322ff6cf2952711a231a7","observation_id":"0a80537c-96f4-40f2-b0c5-5b88f29a12bf","resolution":{"observed_at":"2026-08-06T23:47:38.535627Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-04T10:28:00.554323Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-08-06T20:39:36.782313Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02200","last_updated":"2025-07-02T23:41:31Z","snapshot_observed_at":"2026-08-06T20:32:41.929110Z","submitted_at":"2025-07-02T23:41:31Z","title":"ESTR-CoT: Towards Explainable and Accurate Event Stream based Scene Text Recognition with Chain-of-Thought Reasoning","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T20:39:36.782313Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2507.02200"},"observation_digest":"sha256:eb2a19e4f385aad566d7504f04b35cdace10445ed59dbb0ceb099684f06ded95","observation_id":"cdb269cb-b2e4-4c84-8496-dac5bfbdaa65","resolution":{"observed_at":"2026-08-06T20:39:36.782313Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-04T10:28:00.554323Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-08-06T19:25:02.762351Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.06272","last_updated":"2025-08-09T05:40:33Z","snapshot_observed_at":"2026-08-06T19:16:35.536045Z","submitted_at":"2025-07-08T07:46:26Z","title":"LIRA: Inferring Segmentation in Large Multi-modal Models with Local Interleaved Region Assistance","version":3},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T19:25:02.762351Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2507.06272"},"observation_digest":"sha256:385b03219b31d043253dc6ef1c290517b0ff578abd588c80dacf870eb9a97ea3","observation_id":"0a905ad9-f29c-4e9c-b2ec-4f0b1b862264","resolution":{"observed_at":"2026-08-06T19:25:02.762351Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-04T10:28:00.554323Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-08-06T18:43:18.671551Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.07572","last_updated":"2025-07-10T09:18:06Z","snapshot_observed_at":"2026-08-06T18:35:13.554849Z","submitted_at":"2025-07-10T09:18:06Z","title":"Single-to-mix Modality Alignment with Multimodal Large Language Model for Document Image Machine Translation","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:18.671551Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2507.07572"},"observation_digest":"sha256:38e66401452e1e47050ee4fb8b693dd8379eb83f6c4480c8c1275cae880c999b","observation_id":"836f1d85-7cb8-4389-ad5a-7a2d6669fde8","resolution":{"observed_at":"2026-08-06T18:43:18.671551Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-04T10:28:00.554323Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-08-06T18:28:11.824348Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.08309","last_updated":"2025-07-11T05:02:06Z","snapshot_observed_at":"2026-08-06T18:20:08.589298Z","submitted_at":"2025-07-11T05:02:06Z","title":"Improving MLLM's Document Image Machine Translation via Synchronously Self-reviewing Its OCR Proficiency","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-06T18:28:11.824348Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2507.08309"},"observation_digest":"sha256:9ab48a0e0fe10ebfb72cc90a4f68848ec502164d7c7fa25a3ca94fc7de245908","observation_id":"2195bfaf-1295-49b4-9434-3315e423845d","resolution":{"observed_at":"2026-08-06T18:28:11.824348Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-04T10:28:00.554323Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-08-06T17:59:05.251861Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.09531","last_updated":"2025-07-13T08:15:11Z","snapshot_observed_at":"2026-08-06T17:51:10.172907Z","submitted_at":"2025-07-13T08:15:11Z","title":"VDInstruct: Zero-Shot Key Information Extraction via Content-Aware Vision Tokenization","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:05.251861Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2507.09531"},"observation_digest":"sha256:e3c6ede40bcf98893aa2d27709eafe44da39645223165c5ec5c31ace643935cc","observation_id":"cc6f6b2f-255c-40f5-b1df-25125e92583b","resolution":{"observed_at":"2026-08-06T17:59:05.251861Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-04T10:28:00.554323Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":"2403.04473","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":"612fe156-781e-49f3-bac5-03fcb13d115f","year":2024},"citing_paper":{"arxiv_id":"2507.09861","last_updated":"2026-04-21T13:31:05Z","snapshot_observed_at":"2026-07-06T21:56:35.665677Z","submitted_at":"2025-07-14T02:10:31Z","title":"A Survey on MLLM-based Visually Rich Document Understanding: Methods, Challenges, and Emerging Trends","version":2},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-05-19T04:38:49.512293Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2507.09861"},"observation_digest":"sha256:2aa6864ccd60d14e174b2b268996bb210f8ad8ed5d5cc364217392f4d31bbeef","observation_id":"146b0fc2-3f5b-499c-9f4e-1565187e2bd5","resolution":{"observed_at":"2026-05-19T04:42:04.414900Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-04T10:28:00.554323Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-08-06T16:51:25.051188Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.12441","last_updated":"2025-08-02T17:35:59Z","snapshot_observed_at":"2026-08-06T16:43:19.836401Z","submitted_at":"2025-07-16T17:28:19Z","title":"Describe Anything Model for Visual Question Answering on Text-rich Images","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T16:51:25.051188Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2507.12441"},"observation_digest":"sha256:30bdd9a31f9dd4925eab4ae2bc7dc81898a040e62329e73cc843224f04346eb8","observation_id":"b2244a65-2661-4e5a-9c3e-68e7cf52ac17","resolution":{"observed_at":"2026-08-06T16:51:25.051188Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-04T10:28:00.554323Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-08-06T15:57:02.725020Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.14675","last_updated":"2025-07-19T16:03:34Z","snapshot_observed_at":"2026-08-06T15:47:53.180347Z","submitted_at":"2025-07-19T16:03:34Z","title":"Docopilot: Improving Multimodal Models for Document-Level Understanding","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T15:57:02.725020Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2507.14675"},"observation_digest":"sha256:1fc37e5b6764bbeda9f3d72ae52e977a6f9c1a82328109786e44a32fc95f1687","observation_id":"ece6fc9a-60b6-4784-89f6-ca02fa63b724","resolution":{"observed_at":"2026-08-06T15:57:02.725020Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-04T10:28:00.554323Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-08-05T14:42:32.710762Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document.arXiv preprint arXiv:2403.04473, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.21046","last_updated":"2026-05-27T08:39:30Z","snapshot_observed_at":"2026-08-05T14:42:31.889948Z","submitted_at":"2025-08-28T17:50:58Z","title":"CogVLA: Cognition-Aligned Vision-Language-Action Model via Instruction-Driven Routing & Sparsification","version":3},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-05T14:42:32.710762Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2508.21046"},"observation_digest":"sha256:d087ad851e6c0ef84750a4fa47d906e0528a0f4efb6e6cbc850b9640244674b2","observation_id":"1a118deb-9a5b-4c1f-b5ab-fbba38a6e8da","resolution":{"observed_at":"2026-08-05T14:42:32.710762Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-04T10:28:00.554323Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":"2403.04473","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":"612fe156-781e-49f3-bac5-03fcb13d115f","year":2024},"citing_paper":{"arxiv_id":"2509.22186","last_updated":"2025-09-29T16:41:28Z","snapshot_observed_at":"2026-08-06T11:22:47.957500Z","submitted_at":"2025-09-26T10:45:48Z","title":"MinerU2.5: A Decoupled Vision-Language Model for Efficient High-Resolution Document Parsing","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-17T13:25:31.884175Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2509.22186"},"observation_digest":"sha256:2542afd18bc0e100b90a78fc6c3a8718a12ffe94f14cc060ae22966d57a5730f","observation_id":"b1e09e38-907f-449a-b072-d1c13ad8f18f","resolution":{"observed_at":"2026-05-17T13:25:32.017809Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-04T10:28:00.554323Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-08-03T14:16:50.788856Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document.arXivpreprintarXiv:2403.04473, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2512.21095","last_updated":"2026-07-11T09:04:48Z","snapshot_observed_at":"2026-08-03T14:16:48.399963Z","submitted_at":"2025-12-24T10:35:21Z","title":"UniRec-0.1B: Unified Text and Formula Recognition with 0.1B Parameters","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-03T14:16:50.788856Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2512.21095"},"observation_digest":"sha256:aad12cd4411326eda414321b63251b756e8e90c5b1694a415a58ece403a711a2","observation_id":"97769deb-fd86-43e5-8f63-1cee51cff3d4","resolution":{"observed_at":"2026-08-03T14:16:50.788856Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-04T10:28:00.554323Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-08-02T20:19:14.895125Z","title":"CoRRabs/2403.04473 (2024) 4","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.23615","last_updated":"2026-07-08T09:59:15Z","snapshot_observed_at":"2026-08-04T03:41:58.938743Z","submitted_at":"2026-02-27T02:43:35Z","title":"HART: High-Resolution Annotation-Free Reasoning Technique through a Closed-loop Framework","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-02T20:19:14.895125Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2602.23615"},"observation_digest":"sha256:28c83a342544267c0f21846a7c6f9edf77bcd9ee8ac24e52b261a5bdcd538b4a","observation_id":"a2564aa5-037b-4aff-a5fc-20746eea6fa1","resolution":{"observed_at":"2026-08-02T20:19:14.895125Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-04T10:28:00.554323Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":"2403.04473","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":"612fe156-781e-49f3-bac5-03fcb13d115f","year":2024},"citing_paper":{"arxiv_id":"2603.18472","last_updated":"2026-04-09T02:35:56Z","snapshot_observed_at":"2026-07-06T22:49:37.944352Z","submitted_at":"2026-03-19T04:08:20Z","title":"Cognitive Mismatch in Multimodal Large Language Models for Discrete Symbol Understanding","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-15T09:11:31.870441Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2603.18472"},"observation_digest":"sha256:b5966acade68a1ae0100a2a11db012ea402db24868952356fd3b2231c916661d","observation_id":"1e380ce8-469e-41cf-ad6b-d6c707f8fde4","resolution":{"observed_at":"2026-05-15T09:15:21.113889Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-04T10:28:00.554323Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":"2403.04473","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":"612fe156-781e-49f3-bac5-03fcb13d115f","year":2024},"citing_paper":{"arxiv_id":"2604.00161","last_updated":"2026-04-21T01:45:08Z","snapshot_observed_at":"2026-07-06T22:51:22.254518Z","submitted_at":"2026-03-31T19:09:55Z","title":"Q-Mask: Query-driven Causal Masks for Text Anchoring in OCR-Oriented Vision-Language Models","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:53.449935Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2604.00161"},"observation_digest":"sha256:8d252d8045d97a26da8803e56975263ed237d796a11f4a844525cc1e39b4ea47","observation_id":"44327a09-f177-4137-8c73-f66c3ad7f413","resolution":{"observed_at":"2026-05-13T23:33:26.702294Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-04T10:28:00.554323Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":"2403.04473","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":"612fe156-781e-49f3-bac5-03fcb13d115f","year":2024},"citing_paper":{"arxiv_id":"2604.07419","last_updated":"2026-04-08T14:47:27Z","snapshot_observed_at":"2026-08-02T16:37:23.824907Z","submitted_at":"2026-04-08T14:47:27Z","title":"ReAlign: Optimizing the Visual Document Retriever with Reasoning-Guided Fine-Grained Alignment","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-10T17:43:15.630570Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2604.07419"},"observation_digest":"sha256:b8c8ec92f57659f29971f65a8e52cfcbf46da0cb7fcd30f7f463bbbff781782e","observation_id":"e2959da3-271d-403f-a559-ddea62488bf4","resolution":{"observed_at":"2026-05-11T06:15:58.250374Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-04T10:28:00.554323Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":"2403.04473","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":"612fe156-781e-49f3-bac5-03fcb13d115f","year":2024},"citing_paper":{"arxiv_id":"2604.12812","last_updated":"2026-05-11T03:47:15Z","snapshot_observed_at":"2026-07-06T23:00:56.449281Z","submitted_at":"2026-04-14T14:39:26Z","title":"DocSeeker: Structured Visual Reasoning with Evidence Grounding for Long Document Understanding","version":4},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-10T16:16:58.889065Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2604.12812"},"observation_digest":"sha256:5359b409ae7d99453e00ba752ebe2211087ce037fdf34335da63dd83154fa002","observation_id":"ddd1ef0b-72b4-4ab3-999f-4f7910dcaa41","resolution":{"observed_at":"2026-05-11T09:05:58.472624Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-04T10:28:00.554323Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":"2403.04473","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":"612fe156-781e-49f3-bac5-03fcb13d115f","year":2024},"citing_paper":{"arxiv_id":"2604.12812","last_updated":"2026-05-11T03:47:15Z","snapshot_observed_at":"2026-07-06T23:00:56.449281Z","submitted_at":"2026-04-14T14:39:26Z","title":"DocSeeker: Structured Visual Reasoning with Evidence Grounding for Long Document Understanding","version":5},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-12T04:17:55.318813Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2604.12812"},"observation_digest":"sha256:a9bea269850aeb387f018ce27fba64e920eef7aa237cc1dff00aa00da8aa0be0","observation_id":"16a8f0b1-41c3-4e79-8933-f7731feeafea","resolution":{"observed_at":"2026-05-12T06:26:24.600822Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2403.04473/citation-record","integrity":"/paper/2403.04473/integrity","json":"/paper/2403.04473/citation-record.json","paper":"/paper/2403.04473"},"outbound":[],"paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-04T10:28:00.554323Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 30 inbound Pith citation observations for arXiv:2403.04473."}