{"as_of":"2026-08-04T17:45:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:5d222733be245c75d0c6ded3391771c7f0709936ef5f02734e3e211f0743ebcf","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":26,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":26,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-04T06:34:03.388597+00:00","state":"measured"},{"denominator":26,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":26,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T04:46:26.190461Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-10T06:15:00.866473Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2307.02499","last_updated":"2023-07-04T11:28:07Z","snapshot_observed_at":"2026-08-01T15:24:02.956161Z","submitted_at":"2023-07-04T11:28:07Z","title":"mPLUG-DocOwl: Modularized Multimodal Large Language Model for Document Understanding","version":1},"cited_work":{"arxiv_id":"2307.02499","doi":"10.48550/arxiv.2307.02499","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.02499","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mplug-docowl: Modularized multimodal large language model for document understanding","venue":null,"work_id":"990d95b3-f658-488d-b1a6-169e9e6aa7fb","year":2023},"citing_paper":{"arxiv_id":"2306.13549","last_updated":"2024-11-29T15:51:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-23T15:21:52Z","title":"A Survey on Multimodal Large Language Models","version":4},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-16T02:56:41.658658Z"},"links":{"cited_paper":"/paper/2307.02499","citing_paper":"/paper/2306.13549"},"observation_digest":"sha256:f0dd8968b424e898d58e820a90cd31be33495be7f60e8ba8e6f100607a63973a","observation_id":"552b7a20-9103-4138-af8e-9e7ac74207fe","resolution":{"observed_at":"2026-05-16T02:56:42.407056Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.02499","last_updated":"2023-07-04T11:28:07Z","snapshot_observed_at":"2026-08-01T15:24:02.956161Z","submitted_at":"2023-07-04T11:28:07Z","title":"mPLUG-DocOwl: Modularized Multimodal Large Language Model for Document Understanding","version":1},"cited_work":{"arxiv_id":"2307.02499","doi":"10.48550/arxiv.2307.02499","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.02499","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mplug-docowl: Modularized multimodal large language model for document understanding","venue":null,"work_id":"990d95b3-f658-488d-b1a6-169e9e6aa7fb","year":2023},"citing_paper":{"arxiv_id":"2311.04257","last_updated":"2023-11-09T01:56:51Z","snapshot_observed_at":"2026-07-06T16:44:23.357719Z","submitted_at":"2023-11-07T14:21:29Z","title":"mPLUG-Owl2: Revolutionizing Multi-modal Large Language Model with Modality Collaboration","version":2},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-05-18T03:18:51.582340Z"},"links":{"cited_paper":"/paper/2307.02499","citing_paper":"/paper/2311.04257"},"observation_digest":"sha256:1d9b711a19d6f164f586de5aa5229c5abb139a89ba5d446a995c14e096fd6921","observation_id":"d4b8cd34-6934-476d-8aab-f6f555e1387c","resolution":{"observed_at":"2026-05-18T03:18:51.774332Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.02499","last_updated":"2023-07-04T11:28:07Z","snapshot_observed_at":"2026-08-01T15:24:02.956161Z","submitted_at":"2023-07-04T11:28:07Z","title":"mPLUG-DocOwl: Modularized Multimodal Large Language Model for Document Understanding","version":1},"cited_work":{"arxiv_id":"2307.02499","doi":"10.48550/arxiv.2307.02499","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.02499","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mplug-docowl: Modularized multimodal large language model for document understanding","venue":null,"work_id":"990d95b3-f658-488d-b1a6-169e9e6aa7fb","year":2023},"citing_paper":{"arxiv_id":"2403.09611","last_updated":"2024-04-18T18:51:04Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-03-14T17:51:32Z","title":"MM1: Methods, Analysis & Insights from Multimodal LLM Pre-training","version":4},"reference_index":123,"source":"pdf_text","source_observed_at":"2026-05-16T04:09:36.019146Z"},"links":{"cited_paper":"/paper/2307.02499","citing_paper":"/paper/2403.09611"},"observation_digest":"sha256:0981e3e8d5fceba27b2d435b1813dce49efe4c8a5fdf065245d9819c38fff98e","observation_id":"ebf46cb7-67d1-423c-a210-057833a6f6f0","resolution":{"observed_at":"2026-05-16T04:09:36.354084Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.02499","last_updated":"2023-07-04T11:28:07Z","snapshot_observed_at":"2026-08-01T15:24:02.956161Z","submitted_at":"2023-07-04T11:28:07Z","title":"mPLUG-DocOwl: Modularized Multimodal Large Language Model for Document Understanding","version":1},"cited_work":{"arxiv_id":"2307.02499","doi":"10.48550/arxiv.2307.02499","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.02499","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mplug-docowl: Modularized multimodal large language model for document understanding","venue":null,"work_id":"990d95b3-f658-488d-b1a6-169e9e6aa7fb","year":2023},"citing_paper":{"arxiv_id":"2408.04840","last_updated":"2024-08-13T08:10:32Z","snapshot_observed_at":"2026-07-06T18:58:39.334273Z","submitted_at":"2024-08-09T03:25:42Z","title":"mPLUG-Owl3: Towards Long Image-Sequence Understanding in Multi-Modal Large Language Models","version":2},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-05-20T06:20:36.235304Z"},"links":{"cited_paper":"/paper/2307.02499","citing_paper":"/paper/2408.04840"},"observation_digest":"sha256:912eec3f660a45f08c0bf90a092b1d504ef96e674a9b8a169b77079df7321338","observation_id":"1c2852a6-e767-40dd-90b9-a368384dc822","resolution":{"observed_at":"2026-05-20T06:20:36.308851Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.02499","last_updated":"2023-07-04T11:28:07Z","snapshot_observed_at":"2026-08-01T15:24:02.956161Z","submitted_at":"2023-07-04T11:28:07Z","title":"mPLUG-DocOwl: Modularized Multimodal Large Language Model for Document Understanding","version":1},"cited_work":{"arxiv_id":"2307.02499","doi":"10.48550/arxiv.2307.02499","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.02499","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mplug-docowl: Modularized multimodal large language model for document understanding","venue":null,"work_id":"990d95b3-f658-488d-b1a6-169e9e6aa7fb","year":2023},"citing_paper":{"arxiv_id":"2409.01704","last_updated":"2024-09-03T08:41:31Z","snapshot_observed_at":"2026-08-01T15:04:20.648996Z","submitted_at":"2024-09-03T08:41:31Z","title":"General OCR Theory: Towards OCR-2.0 via a Unified End-to-end Model","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-17T20:50:57.814634Z"},"links":{"cited_paper":"/paper/2307.02499","citing_paper":"/paper/2409.01704"},"observation_digest":"sha256:cd8aeb47c0357d941965d837d629d4420c514f868e5d4eca8ad7fc5d993977c6","observation_id":"ca4339bc-d3f9-4dc1-81b3-4fc30fb7f0ce","resolution":{"observed_at":"2026-05-17T20:50:57.899919Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.02499","last_updated":"2023-07-04T11:28:07Z","snapshot_observed_at":"2026-08-01T15:24:02.956161Z","submitted_at":"2023-07-04T11:28:07Z","title":"mPLUG-DocOwl: Modularized Multimodal Large Language Model for Document Understanding","version":1},"cited_work":{"arxiv_id":"2307.02499","doi":"10.48550/arxiv.2307.02499","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.02499","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mplug-docowl: Modularized multimodal large language model for document understanding","venue":null,"work_id":"990d95b3-f658-488d-b1a6-169e9e6aa7fb","year":2023},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"cited_paper":"/paper/2307.02499","citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:38b76e437ff6942287e2f3077bea43bd14378fac070a69e4a285a87743047048","observation_id":"cd5b616a-9db4-4433-8e40-f861a0281528","resolution":{"observed_at":"2026-05-23T19:43:23.728217Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.02499","last_updated":"2023-07-04T11:28:07Z","snapshot_observed_at":"2026-08-01T15:24:02.956161Z","submitted_at":"2023-07-04T11:28:07Z","title":"mPLUG-DocOwl: Modularized Multimodal Large Language Model for Document Understanding","version":1},"cited_work":{"arxiv_id":"2307.02499","doi":"10.48550/arxiv.2307.02499","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.02499","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mplug-docowl: Modularized multimodal large language model for document understanding","venue":null,"work_id":"990d95b3-f658-488d-b1a6-169e9e6aa7fb","year":2023},"citing_paper":{"arxiv_id":"2410.10594","last_updated":"2025-03-02T01:19:51Z","snapshot_observed_at":"2026-08-02T13:24:00.339129Z","submitted_at":"2024-10-14T15:04:18Z","title":"VisRAG: Vision-based Retrieval-augmented Generation on Multi-modality Documents","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-16T15:37:25.781240Z"},"links":{"cited_paper":"/paper/2307.02499","citing_paper":"/paper/2410.10594"},"observation_digest":"sha256:59f9b28c3e01c0230bcf199b10483a67984fa93beb475c06afd56fad04013d20","observation_id":"f82e1aa3-ea2a-405e-8e62-5e676a63ff1d","resolution":{"observed_at":"2026-05-16T15:37:25.863817Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.02499","last_updated":"2023-07-04T11:28:07Z","snapshot_observed_at":"2026-08-01T15:24:02.956161Z","submitted_at":"2023-07-04T11:28:07Z","title":"mPLUG-DocOwl: Modularized Multimodal Large Language Model for Document Understanding","version":1},"cited_work":{"arxiv_id":"2307.02499","doi":"10.48550/arxiv.2307.02499","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.02499","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mplug-docowl: Modularized multimodal large language model for document understanding","venue":null,"work_id":"990d95b3-f658-488d-b1a6-169e9e6aa7fb","year":2023},"citing_paper":{"arxiv_id":"2501.00321","last_updated":"2025-06-05T02:59:05Z","snapshot_observed_at":"2026-08-02T18:54:27.250149Z","submitted_at":"2024-12-31T07:32:35Z","title":"OCRBench v2: An Improved Benchmark for Evaluating Large Multimodal Models on Visual Text Localization and Reasoning","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-17T20:33:26.613927Z"},"links":{"cited_paper":"/paper/2307.02499","citing_paper":"/paper/2501.00321"},"observation_digest":"sha256:43df4dcdd21b02a5e2632d2c40312be65726e2f16b188da88dfb52a302d755a0","observation_id":"fd1ab135-9013-424b-9169-dbc158c19fcc","resolution":{"observed_at":"2026-05-17T20:33:26.778534Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.02499","last_updated":"2023-07-04T11:28:07Z","snapshot_observed_at":"2026-08-01T15:24:02.956161Z","submitted_at":"2023-07-04T11:28:07Z","title":"mPLUG-DocOwl: Modularized Multimodal Large Language Model for Document Understanding","version":1},"cited_work":{"arxiv_id":"2307.02499","doi":"10.48550/arxiv.2307.02499","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.02499","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mplug-docowl: Modularized multimodal large language model for document understanding","venue":null,"work_id":"990d95b3-f658-488d-b1a6-169e9e6aa7fb","year":2023},"citing_paper":{"arxiv_id":"2503.12937","last_updated":"2025-08-04T04:22:09Z","snapshot_observed_at":"2026-07-31T07:01:13.724738Z","submitted_at":"2025-03-17T08:51:44Z","title":"R1-VL: Learning to Reason with Multimodal Large Language Models via Step-wise Group Relative Policy Optimization","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-16T15:04:22.690503Z"},"links":{"cited_paper":"/paper/2307.02499","citing_paper":"/paper/2503.12937"},"observation_digest":"sha256:d840b564bdcdf2e744a5498c8989a5feaba36ff1b0d091c9621a7e6c5d74a4ff","observation_id":"793e0fc3-89bc-46e8-bca9-fba460646b40","resolution":{"observed_at":"2026-05-16T15:04:22.784774Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.02499","last_updated":"2023-07-04T11:28:07Z","snapshot_observed_at":"2026-08-01T15:24:02.956161Z","submitted_at":"2023-07-04T11:28:07Z","title":"mPLUG-DocOwl: Modularized Multimodal Large Language Model for Document Understanding","version":1},"cited_work":{"arxiv_id":"2307.02499","doi":"10.48550/arxiv.2307.02499","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.02499","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mplug-docowl: Modularized multimodal large language model for document understanding","venue":null,"work_id":"990d95b3-f658-488d-b1a6-169e9e6aa7fb","year":2023},"citing_paper":{"arxiv_id":"2503.14075","last_updated":"2026-04-10T01:08:12Z","snapshot_observed_at":"2026-07-06T20:54:36.522651Z","submitted_at":"2025-03-18T09:52:45Z","title":"Growing a Multi-head Twig via Distillation and Reinforcement Learning to Accelerate Large Vision-Language Models","version":3},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-22T23:58:57.819555Z"},"links":{"cited_paper":"/paper/2307.02499","citing_paper":"/paper/2503.14075"},"observation_digest":"sha256:0b82bceb7d6b99b0090d0241e7488cb63d08bde63ee1ae53529ac206dc3e8af9","observation_id":"4b35a914-4d46-41a5-8e7f-974d3018cb85","resolution":{"observed_at":"2026-05-23T00:02:17.769640Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.02499","last_updated":"2023-07-04T11:28:07Z","snapshot_observed_at":"2026-08-01T15:24:02.956161Z","submitted_at":"2023-07-04T11:28:07Z","title":"mPLUG-DocOwl: Modularized Multimodal Large Language Model for Document Understanding","version":1},"cited_work":{"arxiv_id":"2307.02499","doi":"10.48550/arxiv.2307.02499","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.02499","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mplug-docowl: Modularized multimodal large language model for document understanding","venue":null,"work_id":"990d95b3-f658-488d-b1a6-169e9e6aa7fb","year":2023},"citing_paper":{"arxiv_id":"2507.09861","last_updated":"2026-04-21T13:31:05Z","snapshot_observed_at":"2026-07-06T21:56:35.665677Z","submitted_at":"2025-07-14T02:10:31Z","title":"A Survey on MLLM-based Visually Rich Document Understanding: Methods, Challenges, and Emerging Trends","version":2},"reference_index":65,"source":"arxiv_source","source_observed_at":"2026-05-19T04:38:49.512293Z"},"links":{"cited_paper":"/paper/2307.02499","citing_paper":"/paper/2507.09861"},"observation_digest":"sha256:b17a13c02bbb8f4ca039443eefd127f7ac0a633c9193130ed97e22534391fb60","observation_id":"a032bf00-e823-4686-9959-b392ac51eed5","resolution":{"observed_at":"2026-05-19T04:42:04.350623Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.02499","last_updated":"2023-07-04T11:28:07Z","snapshot_observed_at":"2026-08-01T15:24:02.956161Z","submitted_at":"2023-07-04T11:28:07Z","title":"mPLUG-DocOwl: Modularized Multimodal Large Language Model for Document Understanding","version":1},"cited_work":{"arxiv_id":"2307.02499","doi":"10.48550/arxiv.2307.02499","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.02499","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mplug-docowl: Modularized multimodal large language model for document understanding","venue":null,"work_id":"990d95b3-f658-488d-b1a6-169e9e6aa7fb","year":2023},"citing_paper":{"arxiv_id":"2510.15253","last_updated":"2026-04-20T08:38:26Z","snapshot_observed_at":"2026-07-06T22:32:51.756468Z","submitted_at":"2025-10-17T02:33:16Z","title":"Scaling Beyond Context: A Survey of Multimodal Retrieval-Augmented Generation for Document Understanding","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-18T06:54:03.390655Z"},"links":{"cited_paper":"/paper/2307.02499","citing_paper":"/paper/2510.15253"},"observation_digest":"sha256:329eadaf579cecff92517d7b53402cfd18d69b071ff8ccb912a74d7190b6a99f","observation_id":"0deda2f6-0968-456e-a19b-f59e3ea0d9f2","resolution":{"observed_at":"2026-05-18T06:56:01.635607Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.02499","last_updated":"2023-07-04T11:28:07Z","snapshot_observed_at":"2026-08-01T15:24:02.956161Z","submitted_at":"2023-07-04T11:28:07Z","title":"mPLUG-DocOwl: Modularized Multimodal Large Language Model for Document Understanding","version":1},"cited_work":{"arxiv_id":"2307.02499","doi":"10.48550/arxiv.2307.02499","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.02499","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mplug-docowl: Modularized multimodal large language model for document understanding","venue":null,"work_id":"990d95b3-f658-488d-b1a6-169e9e6aa7fb","year":2023},"citing_paper":{"arxiv_id":"2604.02880","last_updated":"2026-04-17T07:24:14Z","snapshot_observed_at":"2026-07-06T22:52:10.923214Z","submitted_at":"2026-04-03T08:44:45Z","title":"InstructTable: Improving Table Structure Recognition Through Instructions","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-13T20:53:57.029294Z"},"links":{"cited_paper":"/paper/2307.02499","citing_paper":"/paper/2604.02880"},"observation_digest":"sha256:37f5ef98a5155911a976ed9cb5d74274e1f0664beb2e95cd4fe4419e62950a1e","observation_id":"d97af87a-dda4-48a4-ab70-dfe2e3584110","resolution":{"observed_at":"2026-05-13T20:58:16.038985Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.02499","last_updated":"2023-07-04T11:28:07Z","snapshot_observed_at":"2026-08-01T15:24:02.956161Z","submitted_at":"2023-07-04T11:28:07Z","title":"mPLUG-DocOwl: Modularized Multimodal Large Language Model for Document Understanding","version":1},"cited_work":{"arxiv_id":"2307.02499","doi":"10.48550/arxiv.2307.02499","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.02499","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mplug-docowl: Modularized multimodal large language model for document understanding","venue":null,"work_id":"990d95b3-f658-488d-b1a6-169e9e6aa7fb","year":2023},"citing_paper":{"arxiv_id":"2604.04380","last_updated":"2026-04-06T03:04:54Z","snapshot_observed_at":"2026-08-02T13:21:32.442907Z","submitted_at":"2026-04-06T03:04:54Z","title":"CPT: Controllable and Editable Design Variations with Language Models","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-10T20:01:39.648794Z"},"links":{"cited_paper":"/paper/2307.02499","citing_paper":"/paper/2604.04380"},"observation_digest":"sha256:04b704044f69a635cf2c9a531910915e7b0d52244949fa1641586c9ca4b5a0a7","observation_id":"55a49441-a1a8-4ad4-9196-273bfdafe020","resolution":{"observed_at":"2026-05-10T22:15:51.622018Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.02499","last_updated":"2023-07-04T11:28:07Z","snapshot_observed_at":"2026-08-01T15:24:02.956161Z","submitted_at":"2023-07-04T11:28:07Z","title":"mPLUG-DocOwl: Modularized Multimodal Large Language Model for Document Understanding","version":1},"cited_work":{"arxiv_id":"2307.02499","doi":"10.48550/arxiv.2307.02499","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.02499","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mplug-docowl: Modularized multimodal large language model for document understanding","venue":null,"work_id":"990d95b3-f658-488d-b1a6-169e9e6aa7fb","year":2023},"citing_paper":{"arxiv_id":"2604.11042","last_updated":"2026-04-13T06:14:20Z","snapshot_observed_at":"2026-07-06T22:59:32.051572Z","submitted_at":"2026-04-13T06:14:20Z","title":"Improving Layout Representation Learning Across Inconsistently Annotated Datasets via Agentic Harmonization","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-10T16:28:16.315767Z"},"links":{"cited_paper":"/paper/2307.02499","citing_paper":"/paper/2604.11042"},"observation_digest":"sha256:4444e170f237450c732efb7bfc3bc0561b5fca04330e3011fc25138419ef0a78","observation_id":"5b04b076-e704-4649-8935-650c895e0e26","resolution":{"observed_at":"2026-05-11T08:50:58.300577Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.02499","last_updated":"2023-07-04T11:28:07Z","snapshot_observed_at":"2026-08-01T15:24:02.956161Z","submitted_at":"2023-07-04T11:28:07Z","title":"mPLUG-DocOwl: Modularized Multimodal Large Language Model for Document Understanding","version":1},"cited_work":{"arxiv_id":"2307.02499","doi":"10.48550/arxiv.2307.02499","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.02499","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mplug-docowl: Modularized multimodal large language model for document understanding","venue":null,"work_id":"990d95b3-f658-488d-b1a6-169e9e6aa7fb","year":2023},"citing_paper":{"arxiv_id":"2604.12812","last_updated":"2026-05-11T03:47:15Z","snapshot_observed_at":"2026-07-06T23:00:56.449281Z","submitted_at":"2026-04-14T14:39:26Z","title":"DocSeeker: Structured Visual Reasoning with Evidence Grounding for Long Document Understanding","version":4},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-10T16:16:58.889065Z"},"links":{"cited_paper":"/paper/2307.02499","citing_paper":"/paper/2604.12812"},"observation_digest":"sha256:85899c281a50c2e8945171e4713f5bc447978479486f3151cfd1ace977eb4c8b","observation_id":"66a90a39-25bf-43da-836c-38ff35a6b6e3","resolution":{"observed_at":"2026-05-11T09:05:58.486108Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.02499","last_updated":"2023-07-04T11:28:07Z","snapshot_observed_at":"2026-08-01T15:24:02.956161Z","submitted_at":"2023-07-04T11:28:07Z","title":"mPLUG-DocOwl: Modularized Multimodal Large Language Model for Document Understanding","version":1},"cited_work":{"arxiv_id":"2307.02499","doi":"10.48550/arxiv.2307.02499","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.02499","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mplug-docowl: Modularized multimodal large language model for document understanding","venue":null,"work_id":"990d95b3-f658-488d-b1a6-169e9e6aa7fb","year":2023},"citing_paper":{"arxiv_id":"2604.12812","last_updated":"2026-05-11T03:47:15Z","snapshot_observed_at":"2026-07-06T23:00:56.449281Z","submitted_at":"2026-04-14T14:39:26Z","title":"DocSeeker: Structured Visual Reasoning with Evidence Grounding for Long Document Understanding","version":5},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-12T04:17:55.318813Z"},"links":{"cited_paper":"/paper/2307.02499","citing_paper":"/paper/2604.12812"},"observation_digest":"sha256:7221f1e42c687b9147e4130a472271f67d3c6e99704951cf6a973397d78deb1a","observation_id":"eeaf4af3-5576-49c6-9670-71318ade87d9","resolution":{"observed_at":"2026-05-12T06:26:24.507739Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.02499","last_updated":"2023-07-04T11:28:07Z","snapshot_observed_at":"2026-08-01T15:24:02.956161Z","submitted_at":"2023-07-04T11:28:07Z","title":"mPLUG-DocOwl: Modularized Multimodal Large Language Model for Document Understanding","version":1},"cited_work":{"arxiv_id":"2307.02499","doi":"10.48550/arxiv.2307.02499","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.02499","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mplug-docowl: Modularized multimodal large language model for document understanding","venue":null,"work_id":"990d95b3-f658-488d-b1a6-169e9e6aa7fb","year":2023},"citing_paper":{"arxiv_id":"2605.24530","last_updated":"2026-05-23T11:48:28Z","snapshot_observed_at":"2026-07-06T23:34:38.098785Z","submitted_at":"2026-05-23T11:48:28Z","title":"Unveil: Unified Visual-Textual Integration and Distillation for Multi-modal Document Retrieval","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-06-30T13:17:04.441743Z"},"links":{"cited_paper":"/paper/2307.02499","citing_paper":"/paper/2605.24530"},"observation_digest":"sha256:5733dfd0f73b65824b526c1253070789e7baa3c891bb10fdbbf84098d2f88e22","observation_id":"e9aa1bb3-8649-49e7-a2b6-482da0b6da85","resolution":{"observed_at":"2026-06-30T13:24:40.416164Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.02499","last_updated":"2023-07-04T11:28:07Z","snapshot_observed_at":"2026-08-01T15:24:02.956161Z","submitted_at":"2023-07-04T11:28:07Z","title":"mPLUG-DocOwl: Modularized Multimodal Large Language Model for Document Understanding","version":1},"cited_work":{"arxiv_id":"2307.02499","doi":"10.48550/arxiv.2307.02499","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.02499","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mplug-docowl: Modularized multimodal large language model for document understanding","venue":null,"work_id":"990d95b3-f658-488d-b1a6-169e9e6aa7fb","year":2023},"citing_paper":{"arxiv_id":"2606.14061","last_updated":"2026-08-03T06:25:34Z","snapshot_observed_at":"2026-08-04T17:35:37.506748Z","submitted_at":"2026-06-12T03:14:40Z","title":"LLM Agents Can See Code Repositories","version":3},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-06-27T05:22:45.210932Z"},"links":{"cited_paper":"/paper/2307.02499","citing_paper":"/paper/2606.14061"},"observation_digest":"sha256:a9bc09bade50cc78015c9b0aeb20b4d11c0ee28161bf62c3bf5d2a268879eb93","observation_id":"260e67b2-aaaa-497b-8ef6-f37e8d9c6d05","resolution":{"observed_at":"2026-06-27T05:30:36.093677Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.02499","last_updated":"2023-07-04T11:28:07Z","snapshot_observed_at":"2026-08-01T15:24:02.956161Z","submitted_at":"2023-07-04T11:28:07Z","title":"mPLUG-DocOwl: Modularized Multimodal Large Language Model for Document Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.02499","snapshot_observed_at":"2026-08-04T04:46:26.190461Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.14061","last_updated":"2026-08-03T06:25:34Z","snapshot_observed_at":"2026-08-04T17:35:37.506748Z","submitted_at":"2026-06-12T03:14:40Z","title":"LLM Agents Can See Code Repositories","version":4},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-04T04:46:26.190461Z"},"links":{"cited_paper":"/paper/2307.02499","citing_paper":"/paper/2606.14061"},"observation_digest":"sha256:c756f25e578da4770d83403d157f7d5f820f959450a3baf82ee9f864aa63ea3f","observation_id":"65af3fe7-43ec-47db-8b30-253a1489eb4d","resolution":{"observed_at":"2026-08-04T04:46:26.190461Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.02499","last_updated":"2023-07-04T11:28:07Z","snapshot_observed_at":"2026-08-01T15:24:02.956161Z","submitted_at":"2023-07-04T11:28:07Z","title":"mPLUG-DocOwl: Modularized Multimodal Large Language Model for Document Understanding","version":1},"cited_work":{"arxiv_id":"2307.02499","doi":"10.48550/arxiv.2307.02499","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.02499","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mplug-docowl: Modularized multimodal large language model for document understanding","venue":null,"work_id":"990d95b3-f658-488d-b1a6-169e9e6aa7fb","year":2023},"citing_paper":{"arxiv_id":"2606.17030","last_updated":"2026-06-17T13:54:57Z","snapshot_observed_at":"2026-08-01T21:48:31.832288Z","submitted_at":"2026-06-15T17:52:31Z","title":"Qwen-RobotWorld Technical Report: Unifying Embodied World Modeling through Language-Conditioned Video Generation","version":3},"reference_index":99,"source":"arxiv_source","source_observed_at":"2026-06-27T04:19:26.332718Z"},"links":{"cited_paper":"/paper/2307.02499","citing_paper":"/paper/2606.17030"},"observation_digest":"sha256:b61d2a11da5a6deabd5dfd2dc74a5cad13d75b784ceb198f0caf9597f53ac49b","observation_id":"3bbb07e4-ee82-4607-9762-d4ee6dd4254a","resolution":{"observed_at":"2026-07-03T17:18:44.032383Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.02499","last_updated":"2023-07-04T11:28:07Z","snapshot_observed_at":"2026-08-01T15:24:02.956161Z","submitted_at":"2023-07-04T11:28:07Z","title":"mPLUG-DocOwl: Modularized Multimodal Large Language Model for Document Understanding","version":1},"cited_work":{"arxiv_id":"2307.02499","doi":"10.48550/arxiv.2307.02499","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.02499","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mplug-docowl: Modularized multimodal large language model for document understanding","venue":null,"work_id":"990d95b3-f658-488d-b1a6-169e9e6aa7fb","year":2023},"citing_paper":{"arxiv_id":"2606.25343","last_updated":"2026-06-26T04:35:35Z","snapshot_observed_at":"2026-08-03T05:52:16.766463Z","submitted_at":"2026-06-24T03:17:30Z","title":"Invoice Haystack: Benchmarking Document Retrieval and Visual Question Answering Under Strong Visual Homogeneity","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-06-25T21:15:12.242955Z"},"links":{"cited_paper":"/paper/2307.02499","citing_paper":"/paper/2606.25343"},"observation_digest":"sha256:f5356197317492fd85842f8fc9f3bc36ae8df029b80abdaf675d826541f88a06","observation_id":"c091df3d-a442-4ea3-82ba-d232b2938d98","resolution":{"observed_at":"2026-07-04T19:30:07.783753Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.02499","last_updated":"2023-07-04T11:28:07Z","snapshot_observed_at":"2026-08-01T15:24:02.956161Z","submitted_at":"2023-07-04T11:28:07Z","title":"mPLUG-DocOwl: Modularized Multimodal Large Language Model for Document Understanding","version":1},"cited_work":{"arxiv_id":"2307.02499","doi":"10.48550/arxiv.2307.02499","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.02499","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"mplug-docowl: Modularized multimodal large language model for document understanding","venue":null,"work_id":"990d95b3-f658-488d-b1a6-169e9e6aa7fb","year":2023},"citing_paper":{"arxiv_id":"2606.25343","last_updated":"2026-06-26T04:35:35Z","snapshot_observed_at":"2026-08-03T05:52:16.766463Z","submitted_at":"2026-06-24T03:17:30Z","title":"Invoice Haystack: Benchmarking Document Retrieval and Visual Question Answering Under Strong Visual Homogeneity","version":2},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-06-29T05:07:28.537679Z"},"links":{"cited_paper":"/paper/2307.02499","citing_paper":"/paper/2606.25343"},"observation_digest":"sha256:9e474362b75b2a81fea97eff975d458374b86a45704ddd4c2c15673e5d626fd9","observation_id":"66a0e3c9-5836-4a18-908b-d3b9bdbd9293","resolution":{"observed_at":"2026-06-29T18:23:51.235675Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.02499","last_updated":"2023-07-04T11:28:07Z","snapshot_observed_at":"2026-08-01T15:24:02.956161Z","submitted_at":"2023-07-04T11:28:07Z","title":"mPLUG-DocOwl: Modularized Multimodal Large Language Model for Document Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.02499","snapshot_observed_at":"2026-07-14T03:31:19.309532Z","title":"arXiv:2307.02499 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.11738","last_updated":"2026-07-13T16:00:03Z","snapshot_observed_at":"2026-08-02T08:52:21.513313Z","submitted_at":"2026-07-13T16:00:03Z","title":"Qwen-Audio-VAE Technical Report","version":1},"reference_index":114,"source":"arxiv_source","source_observed_at":"2026-07-14T03:31:19.309532Z"},"links":{"cited_paper":"/paper/2307.02499","citing_paper":"/paper/2607.11738"},"observation_digest":"sha256:fee59ff79fc8ba69fa95998750211e4f1cfdebc30dba28edd57933e27505805e","observation_id":"28144c9c-5287-4997-b2ff-b722feec3549","resolution":{"observed_at":"2026-07-14T03:31:19.309532Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.02499","last_updated":"2023-07-04T11:28:07Z","snapshot_observed_at":"2026-08-01T15:24:02.956161Z","submitted_at":"2023-07-04T11:28:07Z","title":"mPLUG-DocOwl: Modularized Multimodal Large Language Model for Document Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.02499","snapshot_observed_at":"2026-07-31T11:50:16.975786Z","title":"mPLUG-DocOwl: Modularized multimodal large language model for document understanding,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.24554","last_updated":"2026-07-27T15:28:02Z","snapshot_observed_at":"2026-08-04T16:42:39.108292Z","submitted_at":"2026-07-27T15:28:02Z","title":"DeCoRAG: Cognitive Decoupling and Semantic-Aware Cropping for Complex Document Understanding","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-07-31T11:50:16.975786Z"},"links":{"cited_paper":"/paper/2307.02499","citing_paper":"/paper/2607.24554"},"observation_digest":"sha256:36bd4fcd41a5f7b77911b0c3f038c912ed9e0dbeaf0ca9548cacbde4e1d655ba","observation_id":"dd04d064-6e10-4f74-8517-0dc5168f0875","resolution":{"observed_at":"2026-07-31T11:50:16.975786Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.02499","last_updated":"2023-07-04T11:28:07Z","snapshot_observed_at":"2026-08-01T15:24:02.956161Z","submitted_at":"2023-07-04T11:28:07Z","title":"mPLUG-DocOwl: Modularized Multimodal Large Language Model for Document Understanding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.02499","snapshot_observed_at":"2026-08-04T01:39:07.816209Z","title":"mplug-docowl: Modularized multimodal large language model for document understanding.arXiv preprint arXiv:2307.02499, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.00036","last_updated":"2026-07-21T09:38:24Z","snapshot_observed_at":"2026-08-04T17:22:46.243468Z","submitted_at":"2026-07-21T09:38:24Z","title":"XL-DocBench: Benchmarking Evidence-Grounded Extra-Long Document Understanding","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-04T01:39:07.816209Z"},"links":{"cited_paper":"/paper/2307.02499","citing_paper":"/paper/2608.00036"},"observation_digest":"sha256:064d90882ba547c7e93cc185497c88c5cce6da498841a0bccb5129d889ea36fe","observation_id":"5f55687b-0776-4368-9f17-49dfe5daaebf","resolution":{"observed_at":"2026-08-04T01:39:07.816209Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2307.02499/citation-record","integrity":"/paper/2307.02499/integrity","json":"/paper/2307.02499/citation-record.json","paper":"/paper/2307.02499"},"outbound":[],"paper":{"arxiv_id":"2307.02499","last_updated":"2023-07-04T11:28:07Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-01T15:24:02.956161Z","submitted_at":"2023-07-04T11:28:07Z","title":"mPLUG-DocOwl: Modularized Multimodal Large Language Model for Document Understanding"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"thesis":"As of 4 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 26 inbound Pith citation observations for arXiv:2307.02499."}