{"as_of":"2026-08-07T09:19:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:48aa75783423147fb5f21d0f60651c61730307ff12019c848196a55e8abd7d83","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":26,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":26,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":26,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":26,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T06:02:30.234729Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-10T17:07:25.743722Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":"2403.12895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-07-10T17:07:25.743722Z","title":"arXiv preprint (2024) DOI: 10.48550/ arXiv.2403.12895","venue":"cs.CV","work_id":"89b0a3a7-ff7a-49bf-af40-a8cf7ee77b16","year":2024},"citing_paper":{"arxiv_id":"2306.13549","last_updated":"2024-11-29T15:51:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-23T15:21:52Z","title":"A Survey on Multimodal Large Language Models","version":4},"reference_index":158,"source":"pdf_text","source_observed_at":"2026-05-16T02:56:41.658658Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2306.13549"},"observation_digest":"sha256:abd1e0f6f85295e56f9ca1ac4af61afadedd80897f98fd11354a7d6175addc18","observation_id":"9636f990-2707-4fc7-ac5b-fcd8b73293c8","resolution":{"observed_at":"2026-05-16T02:56:41.782309Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":"2403.12895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-07-10T17:07:25.743722Z","title":"arXiv preprint (2024) DOI: 10.48550/ arXiv.2403.12895","venue":"cs.CV","work_id":"89b0a3a7-ff7a-49bf-af40-a8cf7ee77b16","year":2024},"citing_paper":{"arxiv_id":"2404.16821","last_updated":"2024-04-29T20:24:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-04-25T17:59:19Z","title":"How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-12T20:58:58.849040Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2404.16821"},"observation_digest":"sha256:fe1c6b610deb9dffa44ef62c5db2ab01ee4e4f1980202f0a0e5de1bc9e9f2ee2","observation_id":"87856638-81ac-452d-8d4d-5731764e1050","resolution":{"observed_at":"2026-05-12T20:58:58.997584Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":"2403.12895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-07-10T17:07:25.743722Z","title":"arXiv preprint (2024) DOI: 10.48550/ arXiv.2403.12895","venue":"cs.CV","work_id":"89b0a3a7-ff7a-49bf-af40-a8cf7ee77b16","year":2024},"citing_paper":{"arxiv_id":"2407.03320","last_updated":"2024-07-03T17:59:21Z","snapshot_observed_at":"2026-08-04T22:09:42.241578Z","submitted_at":"2024-07-03T17:59:21Z","title":"InternLM-XComposer-2.5: A Versatile Large Vision Language Model Supporting Long-Contextual Input and Output","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-17T10:46:28.447347Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2407.03320"},"observation_digest":"sha256:c7114d1de2edfe5519933e45d05cddf5c1d72f87b15508f499829f43c7359903","observation_id":"24372ced-89e2-4eb8-9ceb-ec8101d89e10","resolution":{"observed_at":"2026-05-17T10:46:28.806610Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":"2403.12895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-07-10T17:07:25.743722Z","title":"arXiv preprint (2024) DOI: 10.48550/ arXiv.2403.12895","venue":"cs.CV","work_id":"89b0a3a7-ff7a-49bf-af40-a8cf7ee77b16","year":2024},"citing_paper":{"arxiv_id":"2408.13257","last_updated":"2025-02-05T08:44:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-23T17:59:51Z","title":"MME-RealWorld: Could Your Multimodal LLM Challenge High-Resolution Real-World Scenarios that are Difficult for Humans?","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-16T07:59:32.638758Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2408.13257"},"observation_digest":"sha256:699f79bda2a0be1d658819e8873532eb30fc08b64ac7be85b2ccf6a8a4923a2a","observation_id":"20d6da08-e96d-4de7-90c8-56bcb0da7189","resolution":{"observed_at":"2026-05-16T07:59:32.748963Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":"2403.12895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-07-10T17:07:25.743722Z","title":"arXiv preprint (2024) DOI: 10.48550/ arXiv.2403.12895","venue":"cs.CV","work_id":"89b0a3a7-ff7a-49bf-af40-a8cf7ee77b16","year":2024},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-01T15:04:20.648996Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:8b98eaab83aef4ff070d70bab1fb2ed55f9c1e5703ec9cd2608a3b948eb18ade","observation_id":"29680334-4140-4af1-ae06-56f51675dc11","resolution":{"observed_at":"2026-05-17T20:50:57.864055Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":"2403.12895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-07-10T17:07:25.743722Z","title":"arXiv preprint (2024) DOI: 10.48550/ arXiv.2403.12895","venue":"cs.CV","work_id":"89b0a3a7-ff7a-49bf-af40-a8cf7ee77b16","year":2024},"citing_paper":{"arxiv_id":"2409.18839","last_updated":"2024-09-27T15:35:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-27T15:35:15Z","title":"MinerU: An Open-Source Solution for Precise Document Content Extraction","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-16T04:00:25.624430Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2409.18839"},"observation_digest":"sha256:682dbb00b0b53a5f88821772663e8540cb81221abd0c38c49fd6f3a72a797562","observation_id":"c2efe267-4983-441c-a96f-21bb893de652","resolution":{"observed_at":"2026-05-16T04:00:25.807671Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":"2403.12895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-07-10T17:07:25.743722Z","title":"arXiv preprint (2024) DOI: 10.48550/ arXiv.2403.12895","venue":"cs.CV","work_id":"89b0a3a7-ff7a-49bf-af40-a8cf7ee77b16","year":2024},"citing_paper":{"arxiv_id":"2410.17247","last_updated":"2025-02-27T11:16:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-22T17:59:53Z","title":"PyramidDrop: Accelerating Your Large Vision-Language Models via Pyramid Visual Redundancy Reduction","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-15T12:12:14.613620Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2410.17247"},"observation_digest":"sha256:4bdeffebdd592e90329c9112bb060bf303e51f08289a5f11bc9d580bc5f3f249","observation_id":"20b7ed55-3765-49f7-9512-ba4426200797","resolution":{"observed_at":"2026-05-15T12:12:14.657006Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":"2403.12895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-07-10T17:07:25.743722Z","title":"arXiv preprint (2024) DOI: 10.48550/ arXiv.2403.12895","venue":"cs.CV","work_id":"89b0a3a7-ff7a-49bf-af40-a8cf7ee77b16","year":2024},"citing_paper":{"arxiv_id":"2410.21169","last_updated":"2026-04-04T17:04:02Z","snapshot_observed_at":"2026-07-06T19:40:52.844113Z","submitted_at":"2024-10-28T16:11:35Z","title":"Document Parsing Unveiled: Techniques, Challenges, and Prospects for Structured Information Extraction","version":5},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-05-23T19:15:21.695801Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2410.21169"},"observation_digest":"sha256:3e4f4a8e207a72eb159b62a469f41116f51090f2601dd36b1d6871205be4d834","observation_id":"0bb3dfa2-7fa4-495d-bb7d-4f3df067811d","resolution":{"observed_at":"2026-05-23T19:15:47.181515Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":"2403.12895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-07-10T17:07:25.743722Z","title":"arXiv preprint (2024) DOI: 10.48550/ arXiv.2403.12895","venue":"cs.CV","work_id":"89b0a3a7-ff7a-49bf-af40-a8cf7ee77b16","year":2024},"citing_paper":{"arxiv_id":"2412.05271","last_updated":"2025-09-26T12:52:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-06T18:57:08Z","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","version":5},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-05-10T13:23:57.588851Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2412.05271"},"observation_digest":"sha256:8c549ee083a7a78c0ed66c2f388de7292d03219e38e37679b1903dfd773d1df8","observation_id":"36eb7878-7b97-440b-86f7-8d44812fc120","resolution":{"observed_at":"2026-05-10T13:23:57.855416Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":"2403.12895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-07-10T17:07:25.743722Z","title":"arXiv preprint (2024) DOI: 10.48550/ arXiv.2403.12895","venue":"cs.CV","work_id":"89b0a3a7-ff7a-49bf-af40-a8cf7ee77b16","year":2024},"citing_paper":{"arxiv_id":"2501.00321","last_updated":"2025-06-05T02:59:05Z","snapshot_observed_at":"2026-08-02T18:54:27.250149Z","submitted_at":"2024-12-31T07:32:35Z","title":"OCRBench v2: An Improved Benchmark for Evaluating Large Multimodal Models on Visual Text Localization and Reasoning","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-17T20:33:26.613927Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2501.00321"},"observation_digest":"sha256:338892d9027aaa8f34abd049e0757a81d3e0804c62ab1defa1013e0433dfca11","observation_id":"a94e4643-1970-4d16-8da9-39b3ed21684b","resolution":{"observed_at":"2026-05-17T20:33:26.789867Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-08-07T06:02:30.234729Z","title":"mplug-docowl 1.5: Unified structure learning for ocr-free document understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06279","last_updated":"2025-06-06T17:59:06Z","snapshot_observed_at":"2026-08-07T05:54:42.681893Z","submitted_at":"2025-06-06T17:59:06Z","title":"CoMemo: LVLMs Need Image Context with Image Memory","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-07T06:02:30.234729Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2506.06279"},"observation_digest":"sha256:393ec831980c8d035150ab66a35a1d317ee0f0ef3dbf2c18c0e14735e5151a2f","observation_id":"13ce348c-6f4e-4ff0-a75c-36eed2ec4bee","resolution":{"observed_at":"2026-08-07T06:02:30.234729Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-08-06T23:57:23.097635Z","title":"arXiv preprint arXiv:2403.12895 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.15681","last_updated":"2026-06-25T14:33:27Z","snapshot_observed_at":"2026-08-06T23:49:22.433306Z","submitted_at":"2025-06-18T17:59:49Z","title":"GenRecal: Generation after Recalibration from Large to Small Vision-Language Models","version":4},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T23:57:23.097635Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2506.15681"},"observation_digest":"sha256:f6bb2a6ebf1a38388048b890cadb63ce4db0b3e96191d83ec9692b59da33a1e4","observation_id":"aa7ec550-3dc2-4674-a241-f86be58475fe","resolution":{"observed_at":"2026-08-06T23:57:23.097635Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-08-06T23:47:38.065585Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.21600","last_updated":"2025-06-19T07:16:18Z","snapshot_observed_at":"2026-08-06T23:42:13.492339Z","submitted_at":"2025-06-19T07:16:18Z","title":"Structured Attention Matters to Multimodal LLMs in Document Understanding","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-06T23:47:38.065585Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2506.21600"},"observation_digest":"sha256:0ed6f10ab2caeb691788eef11316ef1bdb85dfe448a6db75f12bb3019a003c07","observation_id":"57b3f667-02d7-41d7-844d-fc169ac4be3f","resolution":{"observed_at":"2026-08-06T23:47:38.065585Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-08-06T21:57:08.059391Z","title":"mplug-docowl 1.5: Unified structure learning for ocr-free document understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23009","last_updated":"2025-08-14T01:29:48Z","snapshot_observed_at":"2026-08-07T08:38:29.921912Z","submitted_at":"2025-06-28T20:46:47Z","title":"MusiXQA: Advancing Visual Music Understanding in Multimodal Large Language Models","version":3},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T21:57:08.059391Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2506.23009"},"observation_digest":"sha256:408c2b50064729234fb6cde2f704d256aeebd2eb1bdecb1233e515edfc58e394","observation_id":"29ef0fcd-2bc9-4672-916c-740cfec8368e","resolution":{"observed_at":"2026-08-06T21:57:08.059391Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-08-06T20:39:37.113140Z","title":"mplug-docowl 1.5: Unified structure learning for ocr-free document understanding,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02200","last_updated":"2025-07-02T23:41:31Z","snapshot_observed_at":"2026-08-06T20:32:41.929110Z","submitted_at":"2025-07-02T23:41:31Z","title":"ESTR-CoT: Towards Explainable and Accurate Event Stream based Scene Text Recognition with Chain-of-Thought Reasoning","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T20:39:37.113140Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2507.02200"},"observation_digest":"sha256:6a09e4f0e43df61b5624783239b729defbdea5894a12bb0dc75053957d60473d","observation_id":"28ac8605-6fd0-46f2-bb60-72bcd742971a","resolution":{"observed_at":"2026-08-06T20:39:37.113140Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-08-06T18:43:17.602697Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.07572","last_updated":"2025-07-10T09:18:06Z","snapshot_observed_at":"2026-08-06T18:35:13.554849Z","submitted_at":"2025-07-10T09:18:06Z","title":"Single-to-mix Modality Alignment with Multimodal Large Language Model for Document Image Machine Translation","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-06T18:43:17.602697Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2507.07572"},"observation_digest":"sha256:3d3a00116c36e9547b92956939e84ed3b2365c2b48533a600bd6c7b9406a5aa8","observation_id":"310cdb34-bc96-4d48-ab61-5808c1d8a2b8","resolution":{"observed_at":"2026-08-06T18:43:17.602697Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":"2403.12895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-07-10T17:07:25.743722Z","title":"arXiv preprint (2024) DOI: 10.48550/ arXiv.2403.12895","venue":"cs.CV","work_id":"89b0a3a7-ff7a-49bf-af40-a8cf7ee77b16","year":2024},"citing_paper":{"arxiv_id":"2507.08458","last_updated":"2026-04-14T14:30:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-11T10:02:08Z","title":"A document is worth a structured record: Principled inductive bias design for document recognition","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-19T04:57:35.441758Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2507.08458"},"observation_digest":"sha256:cffba5881c84ede4422d4de8d58b262a297edc77ac54d37df7c095beb85e4646","observation_id":"3cb95d74-63eb-4d1c-bd59-1439d6c85896","resolution":{"observed_at":"2026-05-19T05:02:05.215068Z","resolver_source":"arxiv_id","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-08-06T16:43:49.641010Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.12883","last_updated":"2025-08-13T05:27:53Z","snapshot_observed_at":"2026-08-06T16:33:30.327495Z","submitted_at":"2025-07-17T08:09:31Z","title":"HRSeg: High-Resolution Visual Perception and Enhancement for Reasoning Segmentation","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T16:43:49.641010Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2507.12883"},"observation_digest":"sha256:5cd62a186d22a18e18d4bbad80ab08227666ba398b29469474e1024c92a485ad","observation_id":"2921b240-0064-4586-9758-4014f25367d4","resolution":{"observed_at":"2026-08-06T16:43:49.641010Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-08-04T11:32:14.796214Z","title":"InICASSP 2025-2025 IEEE International Confer- ence on Acoustics, Speech and Signal Processing (ICASSP), pages 1–5","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2510.04514","last_updated":"2026-06-08T20:02:58Z","snapshot_observed_at":"2026-08-04T11:32:07.464063Z","submitted_at":"2025-10-06T06:05:36Z","title":"ChartAgent: A Multimodal Agent for Visually Grounded Reasoning in Complex Chart Question Answering","version":3},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-04T11:32:14.796214Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2510.04514"},"observation_digest":"sha256:1b82384b2ec366bd2e09cfcd7373d80acbbd7f30675dc5cdd5efa10a383b78e2","observation_id":"7dd154c6-d455-4066-8564-f221c43666c4","resolution":{"observed_at":"2026-08-04T11:32:14.796214Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":"2403.12895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-07-10T17:07:25.743722Z","title":"arXiv preprint (2024) DOI: 10.48550/ arXiv.2403.12895","venue":"cs.CV","work_id":"89b0a3a7-ff7a-49bf-af40-a8cf7ee77b16","year":2024},"citing_paper":{"arxiv_id":"2602.01785","last_updated":"2026-04-28T16:05:53Z","snapshot_observed_at":"2026-07-06T22:44:04.951815Z","submitted_at":"2026-02-02T08:10:21Z","title":"CodeOCR: On the Effectiveness of Vision Language Models in Code Understanding","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-16T08:30:50.984873Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2602.01785"},"observation_digest":"sha256:85f5d78b85183973ea955d8edaf1d6480385c5c18404658f76f6cb2e7679452b","observation_id":"acea7df4-fe20-4d75-9810-9ed67ee89d3c","resolution":{"observed_at":"2026-05-16T08:32:36.576223Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-08-02T20:19:14.038805Z","title":"CoRRabs/2403.12895(2024) 4","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.23615","last_updated":"2026-07-08T09:59:15Z","snapshot_observed_at":"2026-08-04T03:41:58.938743Z","submitted_at":"2026-02-27T02:43:35Z","title":"HART: High-Resolution Annotation-Free Reasoning Technique through a Closed-loop Framework","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-02T20:19:14.038805Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2602.23615"},"observation_digest":"sha256:369162e36ec3390f02212bba17b9eca809a4b46281b9d3add740d89d5f031c9a","observation_id":"dfca068a-ce1c-4348-87f7-f1a55734cadc","resolution":{"observed_at":"2026-08-02T20:19:14.038805Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":"2403.12895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-07-10T17:07:25.743722Z","title":"arXiv preprint (2024) DOI: 10.48550/ arXiv.2403.12895","venue":"cs.CV","work_id":"89b0a3a7-ff7a-49bf-af40-a8cf7ee77b16","year":2024},"citing_paper":{"arxiv_id":"2604.00161","last_updated":"2026-04-21T01:45:08Z","snapshot_observed_at":"2026-07-06T22:51:22.254518Z","submitted_at":"2026-03-31T19:09:55Z","title":"Q-Mask: Query-driven Causal Masks for Text Anchoring in OCR-Oriented Vision-Language Models","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-13T23:30:53.449935Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2604.00161"},"observation_digest":"sha256:36076f55e1c250552af57187ec04eaec322c39ffd40f99999600d899859ff19c","observation_id":"2ba7f8f2-2ea6-4e31-99fd-aa52ca51d8e0","resolution":{"observed_at":"2026-05-13T23:33:26.692458Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":"2403.12895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-07-10T17:07:25.743722Z","title":"arXiv preprint (2024) DOI: 10.48550/ arXiv.2403.12895","venue":"cs.CV","work_id":"89b0a3a7-ff7a-49bf-af40-a8cf7ee77b16","year":2024},"citing_paper":{"arxiv_id":"2604.16883","last_updated":"2026-04-18T07:23:22Z","snapshot_observed_at":"2026-08-03T01:53:31.909118Z","submitted_at":"2026-04-18T07:23:22Z","title":"SinkRouter: Sink-Aware Routing for Efficient Long-Context Decoding in Large Language and Multimodal Models","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-10T07:56:05.583390Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2604.16883"},"observation_digest":"sha256:45eeba2ef99905a202a0820ca43a880f92b224b04273d1b858955b6ecd0cb37d","observation_id":"8ed5f5c4-6e72-483c-9ba1-909a80bff665","resolution":{"observed_at":"2026-05-10T07:57:15.438914Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":"2403.12895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-07-10T17:07:25.743722Z","title":"arXiv preprint (2024) DOI: 10.48550/ arXiv.2403.12895","venue":"cs.CV","work_id":"89b0a3a7-ff7a-49bf-af40-a8cf7ee77b16","year":2024},"citing_paper":{"arxiv_id":"2605.12882","last_updated":"2026-05-13T01:54:42Z","snapshot_observed_at":"2026-07-06T23:24:32.412966Z","submitted_at":"2026-05-13T01:54:42Z","title":"CiteVQA: Benchmarking Evidence Attribution for Trustworthy Document Intelligence","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-14T20:37:36.144960Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2605.12882"},"observation_digest":"sha256:ad3281c65df0b53b5edf4473d58608edc0b65402905dccc8694b22455b2aa2bc","observation_id":"9c6b683e-1544-444d-9dd6-40278a803d47","resolution":{"observed_at":"2026-05-14T20:37:58.163050Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":"2403.12895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-07-10T17:07:25.743722Z","title":"arXiv preprint (2024) DOI: 10.48550/ arXiv.2403.12895","venue":"cs.CV","work_id":"89b0a3a7-ff7a-49bf-af40-a8cf7ee77b16","year":2024},"citing_paper":{"arxiv_id":"2607.07836","last_updated":"2026-07-15T04:21:34Z","snapshot_observed_at":"2026-08-06T18:54:59.892186Z","submitted_at":"2026-07-08T18:17:21Z","title":"Infinity-Parser2 Technical Report","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-07-10T17:02:28.089092Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2607.07836"},"observation_digest":"sha256:1c4c8efeea31c9e1a48e92e5d6aba947b3d8d11065cb8e3a58584b97a0cbcb7a","observation_id":"6492d638-4007-49ba-a24b-3196fe499c0d","resolution":{"observed_at":"2026-07-10T17:07:25.745651Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.12895","snapshot_observed_at":"2026-08-02T08:03:51.775355Z","title":"mPLUG-DocOwl 1.5: Unified structure learning for OCR-free document understanding.arXiv preprint arXiv:2403.12895, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.07836","last_updated":"2026-07-15T04:21:34Z","snapshot_observed_at":"2026-08-06T18:54:59.892186Z","submitted_at":"2026-07-08T18:17:21Z","title":"Infinity-Parser2 Technical Report","version":3},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-02T08:03:51.775355Z"},"links":{"cited_paper":"/paper/2403.12895","citing_paper":"/paper/2607.07836"},"observation_digest":"sha256:05ed70f649b50403c4b6e51c7681518e10d6d2208588e0d8347c89b84a8d3495","observation_id":"ccfee072-f5e6-40a4-be92-bdac0ad074c8","resolution":{"observed_at":"2026-08-02T08:03:51.775355Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2403.12895/citation-record","integrity":"/paper/2403.12895/integrity","json":"/paper/2403.12895/citation-record.json","paper":"/paper/2403.12895"},"outbound":[],"paper":{"arxiv_id":"2403.12895","last_updated":"2024-03-19T16:48:40Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-07T02:52:27.054968Z","submitted_at":"2024-03-19T16:48:40Z","title":"mPLUG-DocOwl 1.5: Unified Structure Learning for OCR-free Document Understanding"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 26 inbound Pith citation observations for arXiv:2403.12895."}