{"as_of":"2026-08-08T05:33:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:cfa720c5590d0c395a944a2446959ecf28e6ce7a0873c75d6271435325d28285","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":38,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":38,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":38,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":38,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T13:53:24.790502Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T21:00:09.476068Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2007.00398","last_updated":"2021-01-05T05:39:39Z","snapshot_observed_at":"2026-07-06T09:34:24.643148Z","submitted_at":"2020-07-01T11:37:40Z","title":"DocVQA: A Dataset for VQA on Document Images","version":3},"cited_work":{"arxiv_id":"2007.00398","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2007.00398","snapshot_observed_at":"2026-07-04T21:00:09.476068Z","title":"Manmatha and C","venue":null,"work_id":"818201c0-eeca-45a8-97ed-9a8c364a749c","year":2007},"citing_paper":{"arxiv_id":"2406.16852","last_updated":"2024-07-01T02:59:29Z","snapshot_observed_at":"2026-08-07T09:52:45.942315Z","submitted_at":"2024-06-24T17:58:06Z","title":"Long Context Transfer from Language to Vision","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-12T07:08:35.946669Z"},"links":{"cited_paper":"/paper/2007.00398","citing_paper":"/paper/2406.16852"},"observation_digest":"sha256:727653c2b2360201da92eda8b05f652224d2da14ceed1318ca2e36d4d1b403f2","observation_id":"9fc75294-2741-4978-a25d-8c0ab778eea9","resolution":{"observed_at":"2026-05-12T07:08:36.126572Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2007.00398","last_updated":"2021-01-05T05:39:39Z","snapshot_observed_at":"2026-07-06T09:34:24.643148Z","submitted_at":"2020-07-01T11:37:40Z","title":"DocVQA: A Dataset for VQA on Document Images","version":3},"cited_work":{"arxiv_id":"2007.00398","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2007.00398","snapshot_observed_at":"2026-07-04T21:00:09.476068Z","title":"Manmatha and C","venue":null,"work_id":"818201c0-eeca-45a8-97ed-9a8c364a749c","year":2007},"citing_paper":{"arxiv_id":"2407.07726","last_updated":"2024-10-10T17:28:23Z","snapshot_observed_at":"2026-08-05T10:06:10.880743Z","submitted_at":"2024-07-10T14:57:46Z","title":"PaliGemma: A versatile 3B VLM for transfer","version":2},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-05-11T13:10:19.972353Z"},"links":{"cited_paper":"/paper/2007.00398","citing_paper":"/paper/2407.07726"},"observation_digest":"sha256:cdd21d743aaf60daecc2f23f6d7becaab72e39e7d1e27f381f601fe9434483ed","observation_id":"50d9352d-b38e-467e-b124-af0163694f8b","resolution":{"observed_at":"2026-05-11T13:10:20.770367Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2007.00398","last_updated":"2021-01-05T05:39:39Z","snapshot_observed_at":"2026-07-06T09:34:24.643148Z","submitted_at":"2020-07-01T11:37:40Z","title":"DocVQA: A Dataset for VQA on Document Images","version":3},"cited_work":{"arxiv_id":"2007.00398","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2007.00398","snapshot_observed_at":"2026-07-04T21:00:09.476068Z","title":"Manmatha and C","venue":null,"work_id":"818201c0-eeca-45a8-97ed-9a8c364a749c","year":2007},"citing_paper":{"arxiv_id":"2412.03555","last_updated":"2024-12-04T18:50:42Z","snapshot_observed_at":"2026-07-06T20:01:45.826971Z","submitted_at":"2024-12-04T18:50:42Z","title":"PaliGemma 2: A Family of Versatile VLMs for Transfer","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-15T09:15:07.523565Z"},"links":{"cited_paper":"/paper/2007.00398","citing_paper":"/paper/2412.03555"},"observation_digest":"sha256:5e607928add17b48f9c33f676fff4b0430a5a580819937610b465237d68a0b5b","observation_id":"db46a353-cc1a-4cdc-8314-d9c82893565f","resolution":{"observed_at":"2026-05-15T09:15:07.712195Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2007.00398","last_updated":"2021-01-05T05:39:39Z","snapshot_observed_at":"2026-07-06T09:34:24.643148Z","submitted_at":"2020-07-01T11:37:40Z","title":"DocVQA: A Dataset for VQA on Document Images","version":3},"cited_work":{"arxiv_id":"2007.00398","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2007.00398","snapshot_observed_at":"2026-07-04T21:00:09.476068Z","title":"Manmatha and C","venue":null,"work_id":"818201c0-eeca-45a8-97ed-9a8c364a749c","year":2007},"citing_paper":{"arxiv_id":"2504.09925","last_updated":"2026-04-29T06:12:36Z","snapshot_observed_at":"2026-08-02T07:57:37.201421Z","submitted_at":"2025-04-14T06:33:29Z","title":"FLARE: Fully Integration of Vision-Language Representations for Deep Cross-Modal Understanding","version":3},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-22T19:49:00.961388Z"},"links":{"cited_paper":"/paper/2007.00398","citing_paper":"/paper/2504.09925"},"observation_digest":"sha256:0ab8f74d3d78ca634d7af23e7957c8e7aa85f82de32fcca9eb784354d415894e","observation_id":"369fbbf5-48c1-4fab-94fc-9a7d3f52ddf4","resolution":{"observed_at":"2026-05-22T19:52:01.796335Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2007.00398","last_updated":"2021-01-05T05:39:39Z","snapshot_observed_at":"2026-07-06T09:34:24.643148Z","submitted_at":"2020-07-01T11:37:40Z","title":"DocVQA: A Dataset for VQA on Document Images","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2007.00398","snapshot_observed_at":"2026-08-07T13:53:24.790502Z","title":"Docvqa: a dataset for vqa on document images","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2505.20753","last_updated":"2025-05-27T05:50:25Z","snapshot_observed_at":"2026-08-07T13:45:07.459562Z","submitted_at":"2025-05-27T05:50:25Z","title":"Understand, Think, and Answer: Advancing Visual Reasoning with Large Multimodal Models","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-07T13:53:24.790502Z"},"links":{"cited_paper":"/paper/2007.00398","citing_paper":"/paper/2505.20753"},"observation_digest":"sha256:11d91f15cac71b867247b7bc4c63718301aae4d97462784fe0af87c0d0b2131e","observation_id":"99fdb9fa-25a6-4dd7-960c-7f37b6558ffc","resolution":{"observed_at":"2026-08-07T13:53:24.790502Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2007.00398","last_updated":"2021-01-05T05:39:39Z","snapshot_observed_at":"2026-07-06T09:34:24.643148Z","submitted_at":"2020-07-01T11:37:40Z","title":"DocVQA: A Dataset for VQA on Document Images","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2007.00398","snapshot_observed_at":"2026-08-07T12:52:39.942212Z","title":"DocVQA: A dataset for VQA on document images","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.23308","last_updated":"2025-05-29T10:06:48Z","snapshot_observed_at":"2026-08-07T21:49:28.223898Z","submitted_at":"2025-05-29T10:06:48Z","title":"Spoken question answering for visual queries","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T12:52:39.942212Z"},"links":{"cited_paper":"/paper/2007.00398","citing_paper":"/paper/2505.23308"},"observation_digest":"sha256:64d26e8b28bc574d7d7b1ec131e0a8db50334d7e1daec2439bccb2b22ee604ee","observation_id":"eab244cd-062d-451b-91ab-aaaca92a01da","resolution":{"observed_at":"2026-08-07T12:52:39.942212Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2007.00398","last_updated":"2021-01-05T05:39:39Z","snapshot_observed_at":"2026-07-06T09:34:24.643148Z","submitted_at":"2020-07-01T11:37:40Z","title":"DocVQA: A Dataset for VQA on Document Images","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2007.00398","snapshot_observed_at":"2026-08-07T12:09:06.849342Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.00479","last_updated":"2025-05-31T09:10:43Z","snapshot_observed_at":"2026-08-07T12:02:00.195241Z","submitted_at":"2025-05-31T09:10:43Z","title":"EffiVLM-BENCH: A Comprehensive Benchmark for Evaluating Training-Free Acceleration in Large Vision-Language Models","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-07T12:09:06.849342Z"},"links":{"cited_paper":"/paper/2007.00398","citing_paper":"/paper/2506.00479"},"observation_digest":"sha256:87658c1bdde4c6c77931b30bea0ae06c6f9fabf68635de5a22138f7b1f9181de","observation_id":"7e1f81cd-a1b2-4d68-b093-2d83107fdba0","resolution":{"observed_at":"2026-08-07T12:09:06.849342Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2007.00398","last_updated":"2021-01-05T05:39:39Z","snapshot_observed_at":"2026-07-06T09:34:24.643148Z","submitted_at":"2020-07-01T11:37:40Z","title":"DocVQA: A Dataset for VQA on Document Images","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2007.00398","snapshot_observed_at":"2026-08-07T10:27:36.789692Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.05182","last_updated":"2025-08-20T20:52:35Z","snapshot_observed_at":"2026-08-07T15:37:22.564913Z","submitted_at":"2025-06-05T15:52:44Z","title":"On the Comprehensibility of Multi-structured Financial Documents using LLMs and Pre-processing Tools","version":2},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-07T10:27:36.789692Z"},"links":{"cited_paper":"/paper/2007.00398","citing_paper":"/paper/2506.05182"},"observation_digest":"sha256:ce785febea7ee8b11be496b149a136ab475d0df74a51fbcf70025db9533e3c5c","observation_id":"202d097c-9a67-4c76-85eb-19e2422e6c74","resolution":{"observed_at":"2026-08-07T10:27:36.789692Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2007.00398","last_updated":"2021-01-05T05:39:39Z","snapshot_observed_at":"2026-07-06T09:34:24.643148Z","submitted_at":"2020-07-01T11:37:40Z","title":"DocVQA: A Dataset for VQA on Document Images","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2007.00398","snapshot_observed_at":"2026-08-06T17:59:05.622333Z","title":"Docvqa: a dataset for vqa on docu- ment images","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2507.09531","last_updated":"2025-07-13T08:15:11Z","snapshot_observed_at":"2026-08-06T17:51:10.172907Z","submitted_at":"2025-07-13T08:15:11Z","title":"VDInstruct: Zero-Shot Key Information Extraction via Content-Aware Vision Tokenization","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T17:59:05.622333Z"},"links":{"cited_paper":"/paper/2007.00398","citing_paper":"/paper/2507.09531"},"observation_digest":"sha256:5f2de58572f98203205257f2544d4bb7d419c7e3780fa14ef2fb2f4a66f528de","observation_id":"40e38e9f-cae0-47d4-9c06-2aae302fb3f7","resolution":{"observed_at":"2026-08-06T17:59:05.622333Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2007.00398","last_updated":"2021-01-05T05:39:39Z","snapshot_observed_at":"2026-07-06T09:34:24.643148Z","submitted_at":"2020-07-01T11:37:40Z","title":"DocVQA: A Dataset for VQA on Document Images","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2007.00398","snapshot_observed_at":"2026-08-06T16:39:20.935176Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.15875","last_updated":"2025-07-17T09:05:34Z","snapshot_observed_at":"2026-08-07T23:04:29.954585Z","submitted_at":"2025-07-17T09:05:34Z","title":"Differential Multimodal Transformers","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T16:39:20.935176Z"},"links":{"cited_paper":"/paper/2007.00398","citing_paper":"/paper/2507.15875"},"observation_digest":"sha256:68fe780e68c252ef47c15bb8fb13482e444e596e9856695e583bc3b2442ece1e","observation_id":"d6733d6f-d499-4144-9636-5486db821583","resolution":{"observed_at":"2026-08-06T16:39:20.935176Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2007.00398","last_updated":"2021-01-05T05:39:39Z","snapshot_observed_at":"2026-07-06T09:34:24.643148Z","submitted_at":"2020-07-01T11:37:40Z","title":"DocVQA: A Dataset for VQA on Document Images","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2007.00398","snapshot_observed_at":"2026-08-06T10:14:26.472521Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2508.00356","last_updated":"2025-08-01T06:39:15Z","snapshot_observed_at":"2026-08-06T13:48:27.648409Z","submitted_at":"2025-08-01T06:39:15Z","title":"Analyze-Prompt-Reason: A Collaborative Agent-Based Framework for Multi-Image Vision-Language Reasoning","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T10:14:26.472521Z"},"links":{"cited_paper":"/paper/2007.00398","citing_paper":"/paper/2508.00356"},"observation_digest":"sha256:7e8a6ca9f4373f0fb918b7d6aa0d3a9df06ce54b6a01990c9724c92eedccfe09","observation_id":"b8386a33-20c2-4723-803b-f2ef52d0b13a","resolution":{"observed_at":"2026-08-06T10:14:26.472521Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2007.00398","last_updated":"2021-01-05T05:39:39Z","snapshot_observed_at":"2026-07-06T09:34:24.643148Z","submitted_at":"2020-07-01T11:37:40Z","title":"DocVQA: A Dataset for VQA on Document Images","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2007.00398","snapshot_observed_at":"2026-08-05T11:11:20.832097Z","title":"Mathew, D","venue":null,"work_id":null,"year":2007},"citing_paper":{"arxiv_id":"2509.06994","last_updated":"2025-09-03T05:54:03Z","snapshot_observed_at":"2026-08-08T03:54:24.303922Z","submitted_at":"2025-09-03T05:54:03Z","title":"VLMs-in-the-Wild: Bridging the Gap Between Academic Benchmarks and Enterprise Reality","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-05T11:11:20.832097Z"},"links":{"cited_paper":"/paper/2007.00398","citing_paper":"/paper/2509.06994"},"observation_digest":"sha256:927ee85029868b292e091b68de7c34694cb483c84466da3fdd4cce965a84096e","observation_id":"2b5709b0-a04a-4bd6-87d9-71b46c381cc6","resolution":{"observed_at":"2026-08-05T11:11:20.832097Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2007.00398","last_updated":"2021-01-05T05:39:39Z","snapshot_observed_at":"2026-07-06T09:34:24.643148Z","submitted_at":"2020-07-01T11:37:40Z","title":"DocVQA: A Dataset for VQA on Document Images","version":3},"cited_work":{"arxiv_id":"2007.00398","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2007.00398","snapshot_observed_at":"2026-07-04T21:00:09.476068Z","title":"Manmatha and C","venue":null,"work_id":"818201c0-eeca-45a8-97ed-9a8c364a749c","year":2007},"citing_paper":{"arxiv_id":"2509.07966","last_updated":"2026-04-21T15:48:38Z","snapshot_observed_at":"2026-07-06T22:27:15.482704Z","submitted_at":"2025-09-09T17:52:26Z","title":"Visual-TableQA: Open-Domain Benchmark for Reasoning over Table Images","version":2},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-05-18T17:37:31.837022Z"},"links":{"cited_paper":"/paper/2007.00398","citing_paper":"/paper/2509.07966"},"observation_digest":"sha256:f8b81cde09cf150ddfa138c92945eff9dc98f2877c7a1a49c1bef201d0740c94","observation_id":"054fa4f0-9302-40b0-aa7c-d26a758aa5bf","resolution":{"observed_at":"2026-05-18T17:42:47.586020Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2007.00398","last_updated":"2021-01-05T05:39:39Z","snapshot_observed_at":"2026-07-06T09:34:24.643148Z","submitted_at":"2020-07-01T11:37:40Z","title":"DocVQA: A Dataset for VQA on Document Images","version":3},"cited_work":{"arxiv_id":"2007.00398","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2007.00398","snapshot_observed_at":"2026-07-04T21:00:09.476068Z","title":"Manmatha and C","venue":null,"work_id":"818201c0-eeca-45a8-97ed-9a8c364a749c","year":2007},"citing_paper":{"arxiv_id":"2511.01831","last_updated":"2026-04-06T20:32:02Z","snapshot_observed_at":"2026-08-02T08:19:53.364576Z","submitted_at":"2025-11-03T18:39:32Z","title":"Routing-Based Continual Learning for Multimodal Large Language Models","version":3},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-05-18T00:52:36.700027Z"},"links":{"cited_paper":"/paper/2007.00398","citing_paper":"/paper/2511.01831"},"observation_digest":"sha256:c22191e4a6817f0c837df5960be61156119fee1c4ed1a96bfb660bef7383c22f","observation_id":"ab61b841-c5db-45f1-afeb-b8d1dd3d97dd","resolution":{"observed_at":"2026-05-18T00:55:35.336139Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2007.00398","last_updated":"2021-01-05T05:39:39Z","snapshot_observed_at":"2026-07-06T09:34:24.643148Z","submitted_at":"2020-07-01T11:37:40Z","title":"DocVQA: A Dataset for VQA on Document Images","version":3},"cited_work":{"arxiv_id":"2007.00398","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2007.00398","snapshot_observed_at":"2026-07-04T21:00:09.476068Z","title":"Manmatha and C","venue":null,"work_id":"818201c0-eeca-45a8-97ed-9a8c364a749c","year":2007},"citing_paper":{"arxiv_id":"2511.14998","last_updated":"2026-04-07T03:13:19Z","snapshot_observed_at":"2026-07-06T22:36:13.287872Z","submitted_at":"2025-11-19T00:41:14Z","title":"FinCriticalED: A Visual Benchmark for Financial Fact-Level OCR","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-17T20:19:52.701262Z"},"links":{"cited_paper":"/paper/2007.00398","citing_paper":"/paper/2511.14998"},"observation_digest":"sha256:eb70a753093af927ea76cbd3495054b37b2909dcb914ced1cd8b644e2545bed1","observation_id":"2fdfb66c-d33a-4093-b387-627443e15855","resolution":{"observed_at":"2026-05-17T20:20:11.510920Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2007.00398","last_updated":"2021-01-05T05:39:39Z","snapshot_observed_at":"2026-07-06T09:34:24.643148Z","submitted_at":"2020-07-01T11:37:40Z","title":"DocVQA: A Dataset for VQA on Document Images","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2007.00398","snapshot_observed_at":"2026-08-02T19:41:33.738459Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2603.01471","last_updated":"2026-06-02T03:52:48Z","snapshot_observed_at":"2026-08-07T18:14:02.067221Z","submitted_at":"2026-03-02T05:34:45Z","title":"Reconstructing Content with Collaborative Attention for Universal Multimodal Representation Learning","version":3},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-02T19:41:33.738459Z"},"links":{"cited_paper":"/paper/2007.00398","citing_paper":"/paper/2603.01471"},"observation_digest":"sha256:a0ccdeda9b360106287912cfc346c714a436f5e00fbcded7030218eb234b62d6","observation_id":"2d897ada-8f28-4f91-9800-5f8d6012ed55","resolution":{"observed_at":"2026-08-02T19:41:33.738459Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2007.00398","last_updated":"2021-01-05T05:39:39Z","snapshot_observed_at":"2026-07-06T09:34:24.643148Z","submitted_at":"2020-07-01T11:37:40Z","title":"DocVQA: A Dataset for VQA on Document Images","version":3},"cited_work":{"arxiv_id":"2007.00398","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2007.00398","snapshot_observed_at":"2026-07-04T21:00:09.476068Z","title":"Manmatha and C","venue":null,"work_id":"818201c0-eeca-45a8-97ed-9a8c364a749c","year":2007},"citing_paper":{"arxiv_id":"2604.04901","last_updated":"2026-04-06T17:49:31Z","snapshot_observed_at":"2026-07-06T22:53:46.999911Z","submitted_at":"2026-04-06T17:49:31Z","title":"FileGram: Grounding Agent Personalization in File-System Behavioral Traces","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-10T19:33:57.643549Z"},"links":{"cited_paper":"/paper/2007.00398","citing_paper":"/paper/2604.04901"},"observation_digest":"sha256:efdb4dee63ac6fc7844fb6b66f1480f63822e2e90373c127548848623c182584","observation_id":"e8010136-aa82-4ae3-8ded-81593f01f200","resolution":{"observed_at":"2026-05-10T22:45:51.276505Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2007.00398","last_updated":"2021-01-05T05:39:39Z","snapshot_observed_at":"2026-07-06T09:34:24.643148Z","submitted_at":"2020-07-01T11:37:40Z","title":"DocVQA: A Dataset for VQA on Document Images","version":3},"cited_work":{"arxiv_id":"2007.00398","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2007.00398","snapshot_observed_at":"2026-07-04T21:00:09.476068Z","title":"Manmatha and C","venue":null,"work_id":"818201c0-eeca-45a8-97ed-9a8c364a749c","year":2007},"citing_paper":{"arxiv_id":"2604.06912","last_updated":"2026-04-08T10:12:30Z","snapshot_observed_at":"2026-08-04T18:58:19.981225Z","submitted_at":"2026-04-08T10:12:30Z","title":"Q-Zoom: Query-Aware Adaptive Perception for Efficient Multimodal Large Language Models","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-10T18:46:26.869644Z"},"links":{"cited_paper":"/paper/2007.00398","citing_paper":"/paper/2604.06912"},"observation_digest":"sha256:54d182cf195b6b185382a27887f8dc945c9571bfc3de19f7769060dc1027d136","observation_id":"756994a1-db77-498a-90b5-6873aee623f7","resolution":{"observed_at":"2026-05-10T23:55:52.856348Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2007.00398","last_updated":"2021-01-05T05:39:39Z","snapshot_observed_at":"2026-07-06T09:34:24.643148Z","submitted_at":"2020-07-01T11:37:40Z","title":"DocVQA: A Dataset for VQA on Document Images","version":3},"cited_work":{"arxiv_id":"2007.00398","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2007.00398","snapshot_observed_at":"2026-07-04T21:00:09.476068Z","title":"Manmatha and C","venue":null,"work_id":"818201c0-eeca-45a8-97ed-9a8c364a749c","year":2007},"citing_paper":{"arxiv_id":"2604.08212","last_updated":"2026-04-09T13:11:30Z","snapshot_observed_at":"2026-07-06T22:57:22.475588Z","submitted_at":"2026-04-09T13:11:30Z","title":"Vision-Language Foundation Models for Comprehensive Automated Pavement Condition Assessment","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-10T17:30:39.410040Z"},"links":{"cited_paper":"/paper/2007.00398","citing_paper":"/paper/2604.08212"},"observation_digest":"sha256:672401e4f3bc9dcc886e0f8f189c48f6375ee1b017210dfe61b851f7a04bd3a0","observation_id":"99a7eafb-0ef2-47b6-bbc3-23c10c5fed5c","resolution":{"observed_at":"2026-05-11T06:41:22.377691Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2007.00398","last_updated":"2021-01-05T05:39:39Z","snapshot_observed_at":"2026-07-06T09:34:24.643148Z","submitted_at":"2020-07-01T11:37:40Z","title":"DocVQA: A Dataset for VQA on Document Images","version":3},"cited_work":{"arxiv_id":"2007.00398","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2007.00398","snapshot_observed_at":"2026-07-04T21:00:09.476068Z","title":"Manmatha and C","venue":null,"work_id":"818201c0-eeca-45a8-97ed-9a8c364a749c","year":2007},"citing_paper":{"arxiv_id":"2604.08456","last_updated":"2026-04-09T16:51:42Z","snapshot_observed_at":"2026-07-06T22:57:31.046146Z","submitted_at":"2026-04-09T16:51:42Z","title":"Entropy-Gradient Grounding: Training-Free Evidence Retrieval in Vision-Language Models","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-10T17:14:11.941977Z"},"links":{"cited_paper":"/paper/2007.00398","citing_paper":"/paper/2604.08456"},"observation_digest":"sha256:10e851bdd1d27dc84fec2227fca2c0e96b7e23b2699406993fdb817196a676cf","observation_id":"f54d59ad-4dd6-4343-9bec-42260ac96d19","resolution":{"observed_at":"2026-05-11T07:16:11.146347Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2007.00398","last_updated":"2021-01-05T05:39:39Z","snapshot_observed_at":"2026-07-06T09:34:24.643148Z","submitted_at":"2020-07-01T11:37:40Z","title":"DocVQA: A Dataset for VQA on Document Images","version":3},"cited_work":{"arxiv_id":"2007.00398","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2007.00398","snapshot_observed_at":"2026-07-04T21:00:09.476068Z","title":"Manmatha and C","venue":null,"work_id":"818201c0-eeca-45a8-97ed-9a8c364a749c","year":2007},"citing_paper":{"arxiv_id":"2604.14799","last_updated":"2026-04-16T09:23:22Z","snapshot_observed_at":"2026-08-02T12:46:46.976919Z","submitted_at":"2026-04-16T09:23:22Z","title":"Knowing When Not to Answer: Evaluating Abstention in Multimodal Reasoning Systems","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T11:42:13.100463Z"},"links":{"cited_paper":"/paper/2007.00398","citing_paper":"/paper/2604.14799"},"observation_digest":"sha256:6714fe5aca53332f020a700c1af500e8b6570888126175867d2180d9e6a7536f","observation_id":"ed301277-68ed-493f-b006-79df3acf6942","resolution":{"observed_at":"2026-05-10T11:50:20.766938Z","resolver_source":"arxiv_id","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2007.00398","last_updated":"2021-01-05T05:39:39Z","snapshot_observed_at":"2026-07-06T09:34:24.643148Z","submitted_at":"2020-07-01T11:37:40Z","title":"DocVQA: A Dataset for VQA on Document Images","version":3},"cited_work":{"arxiv_id":"2007.00398","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2007.00398","snapshot_observed_at":"2026-07-04T21:00:09.476068Z","title":"Manmatha and C","venue":null,"work_id":"818201c0-eeca-45a8-97ed-9a8c364a749c","year":2007},"citing_paper":{"arxiv_id":"2604.19503","last_updated":"2026-05-09T05:54:49Z","snapshot_observed_at":"2026-08-06T04:17:17.024078Z","submitted_at":"2026-04-21T14:22:04Z","title":"ReaLB: Real-Time Load Balancing for Multimodal MoE Inference","version":2},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-05-10T01:18:27.597377Z"},"links":{"cited_paper":"/paper/2007.00398","citing_paper":"/paper/2604.19503"},"observation_digest":"sha256:fb04ae7401fea012014cf8726250ad9f53513e6137af90164a4c0d9bce4d35f1","observation_id":"1bb123da-ab2f-465d-98e3-8289bbacef79","resolution":{"observed_at":"2026-05-11T13:36:10.208047Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2007.00398","last_updated":"2021-01-05T05:39:39Z","snapshot_observed_at":"2026-07-06T09:34:24.643148Z","submitted_at":"2020-07-01T11:37:40Z","title":"DocVQA: A Dataset for VQA on Document Images","version":3},"cited_work":{"arxiv_id":"2007.00398","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2007.00398","snapshot_observed_at":"2026-07-04T21:00:09.476068Z","title":"Manmatha and C","venue":null,"work_id":"818201c0-eeca-45a8-97ed-9a8c364a749c","year":2007},"citing_paper":{"arxiv_id":"2604.19945","last_updated":"2026-04-21T19:48:19Z","snapshot_observed_at":"2026-08-02T17:45:19.558396Z","submitted_at":"2026-04-21T19:48:19Z","title":"Visual Reasoning through Tool-supervised Reinforcement Learning","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-10T03:05:21.688216Z"},"links":{"cited_paper":"/paper/2007.00398","citing_paper":"/paper/2604.19945"},"observation_digest":"sha256:ef0a9e86cab42848d1849b86eabdd84797185e3b410155cd90cf17e9e0606d7a","observation_id":"fd9d38bd-1825-406a-be64-8868f184dab1","resolution":{"observed_at":"2026-05-11T12:46:03.231775Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2007.00398","last_updated":"2021-01-05T05:39:39Z","snapshot_observed_at":"2026-07-06T09:34:24.643148Z","submitted_at":"2020-07-01T11:37:40Z","title":"DocVQA: A Dataset for VQA on Document Images","version":3},"cited_work":{"arxiv_id":"2007.00398","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2007.00398","snapshot_observed_at":"2026-07-04T21:00:09.476068Z","title":"Manmatha and C","venue":null,"work_id":"818201c0-eeca-45a8-97ed-9a8c364a749c","year":2007},"citing_paper":{"arxiv_id":"2604.25185","last_updated":"2026-06-18T03:00:25Z","snapshot_observed_at":"2026-08-05T20:03:52.003085Z","submitted_at":"2026-04-28T03:43:53Z","title":"The category of Whittaker modules over the Cartan Type Lie algebra $\\bar{S}_2$","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-07-01T09:08:06.592577Z"},"links":{"cited_paper":"/paper/2007.00398","citing_paper":"/paper/2604.25185"},"observation_digest":"sha256:eb811e95ee4f74d9a9434a6cb4374dcf41a87f424ad1fe3b12575fd5599d589b","observation_id":"38b54dfb-70d2-4d82-9413-4a23a8d5eea4","resolution":{"observed_at":"2026-07-01T09:25:40.475504Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2007.00398","last_updated":"2021-01-05T05:39:39Z","snapshot_observed_at":"2026-07-06T09:34:24.643148Z","submitted_at":"2020-07-01T11:37:40Z","title":"DocVQA: A Dataset for VQA on Document Images","version":3},"cited_work":{"arxiv_id":"2007.00398","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2007.00398","snapshot_observed_at":"2026-07-04T21:00:09.476068Z","title":"Manmatha and C","venue":null,"work_id":"818201c0-eeca-45a8-97ed-9a8c364a749c","year":2007},"citing_paper":{"arxiv_id":"2604.25186","last_updated":"2026-04-30T03:30:43Z","snapshot_observed_at":"2026-07-06T23:11:07.254460Z","submitted_at":"2026-04-28T03:45:09Z","title":"FCMBench-Video: Benchmarking Document Video Intelligence","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-07T17:14:19.186123Z"},"links":{"cited_paper":"/paper/2007.00398","citing_paper":"/paper/2604.25186"},"observation_digest":"sha256:aa46606a1efd7b381752d5e49738212f29598dcaab6c4e39516e65c3cefdf2fc","observation_id":"3b1e86e4-3282-4455-a195-5d9290b8e94a","resolution":{"observed_at":"2026-05-11T23:21:41.346550Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2007.00398","last_updated":"2021-01-05T05:39:39Z","snapshot_observed_at":"2026-07-06T09:34:24.643148Z","submitted_at":"2020-07-01T11:37:40Z","title":"DocVQA: A Dataset for VQA on Document Images","version":3},"cited_work":{"arxiv_id":"2007.00398","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2007.00398","snapshot_observed_at":"2026-07-04T21:00:09.476068Z","title":"Manmatha and C","venue":null,"work_id":"818201c0-eeca-45a8-97ed-9a8c364a749c","year":2007},"citing_paper":{"arxiv_id":"2605.04075","last_updated":"2026-04-14T08:17:53Z","snapshot_observed_at":"2026-08-03T03:08:39.084770Z","submitted_at":"2026-04-14T08:17:53Z","title":"RetentiveKV: State-Space Memory for Uncertainty-Aware Multimodal KV Cache Eviction","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-05-10T15:29:17.567557Z"},"links":{"cited_paper":"/paper/2007.00398","citing_paper":"/paper/2605.04075"},"observation_digest":"sha256:aaa66547d6372d383ee1b7a41a19d883b2fb40024cfbf9f6c3515be6e289a704","observation_id":"fec6371b-9edf-4e6c-b000-4668c7c106e8","resolution":{"observed_at":"2026-05-11T10:26:02.150150Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2007.00398","last_updated":"2021-01-05T05:39:39Z","snapshot_observed_at":"2026-07-06T09:34:24.643148Z","submitted_at":"2020-07-01T11:37:40Z","title":"DocVQA: A Dataset for VQA on Document Images","version":3},"cited_work":{"arxiv_id":"2007.00398","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2007.00398","snapshot_observed_at":"2026-07-04T21:00:09.476068Z","title":"Manmatha and C","venue":null,"work_id":"818201c0-eeca-45a8-97ed-9a8c364a749c","year":2007},"citing_paper":{"arxiv_id":"2605.20950","last_updated":"2026-05-20T09:37:53Z","snapshot_observed_at":"2026-07-06T23:31:27.989406Z","submitted_at":"2026-05-20T09:37:53Z","title":"Focus-then-Context: Subject-Centric Progressive Visual Token Reduction for Vision-Language Models","version":1},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-05-21T05:20:55.448430Z"},"links":{"cited_paper":"/paper/2007.00398","citing_paper":"/paper/2605.20950"},"observation_digest":"sha256:f467575014e79287866b4cabdeea6d6afc7cf2f38508f99f59cf424d69051afb","observation_id":"430876f6-6afd-4c69-92ed-2e3c253887b0","resolution":{"observed_at":"2026-05-21T05:23:58.581700Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2007.00398","last_updated":"2021-01-05T05:39:39Z","snapshot_observed_at":"2026-07-06T09:34:24.643148Z","submitted_at":"2020-07-01T11:37:40Z","title":"DocVQA: A Dataset for VQA on Document Images","version":3},"cited_work":{"arxiv_id":"2007.00398","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2007.00398","snapshot_observed_at":"2026-07-04T21:00:09.476068Z","title":"Manmatha and C","venue":null,"work_id":"818201c0-eeca-45a8-97ed-9a8c364a749c","year":2007},"citing_paper":{"arxiv_id":"2605.30244","last_updated":"2026-05-28T17:11:03Z","snapshot_observed_at":"2026-07-06T23:39:34.232661Z","submitted_at":"2026-05-28T17:11:03Z","title":"Reinforcement Learning with Robust Rubric Rewards","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-06-29T07:39:21.677389Z"},"links":{"cited_paper":"/paper/2007.00398","citing_paper":"/paper/2605.30244"},"observation_digest":"sha256:17c2962462beb888790c85d6355c8824d1470f644a7a571ea960b87dd1cd3bed","observation_id":"fdc13621-692c-4c1a-9d69-acb3c29209ca","resolution":{"observed_at":"2026-06-29T07:43:13.742120Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2007.00398","last_updated":"2021-01-05T05:39:39Z","snapshot_observed_at":"2026-07-06T09:34:24.643148Z","submitted_at":"2020-07-01T11:37:40Z","title":"DocVQA: A Dataset for VQA on Document Images","version":3},"cited_work":{"arxiv_id":"2007.00398","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2007.00398","snapshot_observed_at":"2026-07-04T21:00:09.476068Z","title":"Manmatha and C","venue":null,"work_id":"818201c0-eeca-45a8-97ed-9a8c364a749c","year":2007},"citing_paper":{"arxiv_id":"2605.31604","last_updated":"2026-07-03T09:50:38Z","snapshot_observed_at":"2026-08-03T15:39:37.184662Z","submitted_at":"2026-05-29T17:59:55Z","title":"Representation Forcing for Bottleneck-Free Unified Multimodal Models","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-06-28T22:54:10.460872Z"},"links":{"cited_paper":"/paper/2007.00398","citing_paper":"/paper/2605.31604"},"observation_digest":"sha256:ab086a93cfb8814475032ec0503fd53449954e565c9b99d46b562c7b14b8aba2","observation_id":"cbe0c011-0592-4df5-9f91-8eeaf596ab3f","resolution":{"observed_at":"2026-07-01T19:16:00.977140Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2007.00398","last_updated":"2021-01-05T05:39:39Z","snapshot_observed_at":"2026-07-06T09:34:24.643148Z","submitted_at":"2020-07-01T11:37:40Z","title":"DocVQA: A Dataset for VQA on Document Images","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2007.00398","snapshot_observed_at":"2026-07-12T15:31:57.426559Z","title":null,"venue":null,"work_id":null,"year":2007},"citing_paper":{"arxiv_id":"2605.31604","last_updated":"2026-07-03T09:50:38Z","snapshot_observed_at":"2026-08-03T15:39:37.184662Z","submitted_at":"2026-05-29T17:59:55Z","title":"Representation Forcing for Bottleneck-Free Unified Multimodal Models","version":4},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-07-12T15:31:57.426559Z"},"links":{"cited_paper":"/paper/2007.00398","citing_paper":"/paper/2605.31604"},"observation_digest":"sha256:23caae1703f6935f0c1bcdbea855b825230336774539acb4bfbef398a863528f","observation_id":"4c9a5c00-a365-4c3e-89fc-85f4d9e504a0","resolution":{"observed_at":"2026-07-12T15:31:57.426559Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2007.00398","last_updated":"2021-01-05T05:39:39Z","snapshot_observed_at":"2026-07-06T09:34:24.643148Z","submitted_at":"2020-07-01T11:37:40Z","title":"DocVQA: A Dataset for VQA on Document Images","version":3},"cited_work":{"arxiv_id":"2007.00398","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2007.00398","snapshot_observed_at":"2026-07-04T21:00:09.476068Z","title":"Manmatha and C","venue":null,"work_id":"818201c0-eeca-45a8-97ed-9a8c364a749c","year":2007},"citing_paper":{"arxiv_id":"2606.17110","last_updated":"2026-06-15T07:04:01Z","snapshot_observed_at":"2026-08-05T23:11:05.452983Z","submitted_at":"2026-06-15T07:04:01Z","title":"Loss Landscape Poisoning: Targeted Extraction of Unseen Training Data from LLMs","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-06-27T03:59:30.468854Z"},"links":{"cited_paper":"/paper/2007.00398","citing_paper":"/paper/2606.17110"},"observation_digest":"sha256:96aaf6d3c9d9786ae70a3889956afa06c5ab5b1cb494756d2fd6f80a73fa3f45","observation_id":"010cabf7-d743-4021-b8fc-fcd648a84a74","resolution":{"observed_at":"2026-07-03T17:38:43.814213Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2007.00398","last_updated":"2021-01-05T05:39:39Z","snapshot_observed_at":"2026-07-06T09:34:24.643148Z","submitted_at":"2020-07-01T11:37:40Z","title":"DocVQA: A Dataset for VQA on Document Images","version":3},"cited_work":{"arxiv_id":"2007.00398","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2007.00398","snapshot_observed_at":"2026-07-04T21:00:09.476068Z","title":"Manmatha and C","venue":null,"work_id":"818201c0-eeca-45a8-97ed-9a8c364a749c","year":2007},"citing_paper":{"arxiv_id":"2606.26041","last_updated":"2026-06-24T17:15:42Z","snapshot_observed_at":"2026-08-02T15:35:24.404490Z","submitted_at":"2026-06-24T17:15:42Z","title":"How Robust is OCR-Reasoning? Evaluating OCR-Reasoning Robustness of Vision-Language Models under Visual Perturbations","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-06-25T19:13:27.527971Z"},"links":{"cited_paper":"/paper/2007.00398","citing_paper":"/paper/2606.26041"},"observation_digest":"sha256:49f31951d0fbb06c02df3562ac545cac178a4531067f64598b71b1ba88ae1c20","observation_id":"1bf935f7-a473-45a5-873b-0012da53ba03","resolution":{"observed_at":"2026-07-04T21:00:09.477417Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2007.00398","last_updated":"2021-01-05T05:39:39Z","snapshot_observed_at":"2026-07-06T09:34:24.643148Z","submitted_at":"2020-07-01T11:37:40Z","title":"DocVQA: A Dataset for VQA on Document Images","version":3},"cited_work":{"arxiv_id":"2007.00398","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2007.00398","snapshot_observed_at":"2026-07-04T21:00:09.476068Z","title":"Manmatha and C","venue":null,"work_id":"818201c0-eeca-45a8-97ed-9a8c364a749c","year":2007},"citing_paper":{"arxiv_id":"2607.02484","last_updated":"2026-07-02T17:50:57Z","snapshot_observed_at":"2026-07-07T00:07:56.573640Z","submitted_at":"2026-07-02T17:50:57Z","title":"Combating Textual Noise and Redundancy: Entropy-Aware Dense Visual Token Pruning","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-07-03T14:47:35.377391Z"},"links":{"cited_paper":"/paper/2007.00398","citing_paper":"/paper/2607.02484"},"observation_digest":"sha256:7390d61524c656f2b001ddb828ebd9db5094b4e933950ebe4874750da8783836","observation_id":"d3a070c7-8575-4c1c-96f1-715e82e35265","resolution":{"observed_at":"2026-07-03T14:48:32.412815Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2007.00398","last_updated":"2021-01-05T05:39:39Z","snapshot_observed_at":"2026-07-06T09:34:24.643148Z","submitted_at":"2020-07-01T11:37:40Z","title":"DocVQA: A Dataset for VQA on Document Images","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2007.00398","snapshot_observed_at":"2026-07-12T01:07:20.766474Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2607.03624","last_updated":"2026-07-03T22:58:54Z","snapshot_observed_at":"2026-08-01T21:59:16.066897Z","submitted_at":"2026-07-03T22:58:54Z","title":"RADIO1D: Elastic Representations for Condensed Vision Modeling","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-07-12T01:07:20.766474Z"},"links":{"cited_paper":"/paper/2007.00398","citing_paper":"/paper/2607.03624"},"observation_digest":"sha256:e057774a4fe6d059ee0468ea21a6288aa20f9b6fd23ad91d68184ef72b774765","observation_id":"d0a1ea24-90f5-47ea-a7ef-7e51dccc1a17","resolution":{"observed_at":"2026-07-12T01:07:20.766474Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2007.00398","last_updated":"2021-01-05T05:39:39Z","snapshot_observed_at":"2026-07-06T09:34:24.643148Z","submitted_at":"2020-07-01T11:37:40Z","title":"DocVQA: A Dataset for VQA on Document Images","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2007.00398","snapshot_observed_at":"2026-07-31T06:18:55.901623Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.24744","last_updated":"2026-07-27T17:59:58Z","snapshot_observed_at":"2026-08-06T17:57:37.442251Z","submitted_at":"2026-07-27T17:59:58Z","title":"Data Pyramid for Embodied Manipulation","version":1},"reference_index":253,"source":"pdf_text","source_observed_at":"2026-07-31T06:18:55.901623Z"},"links":{"cited_paper":"/paper/2007.00398","citing_paper":"/paper/2607.24744"},"observation_digest":"sha256:5c26ca5d6588779dee8883105cf11094b37d7dedbb30a0bd65a2ef023d99127d","observation_id":"7e001a6e-0255-4ee9-93d8-2d2459c10395","resolution":{"observed_at":"2026-07-31T06:18:55.901623Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2007.00398","last_updated":"2021-01-05T05:39:39Z","snapshot_observed_at":"2026-07-06T09:34:24.643148Z","submitted_at":"2020-07-01T11:37:40Z","title":"DocVQA: A Dataset for VQA on Document Images","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2007.00398","snapshot_observed_at":"2026-08-02T14:38:35.492534Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.24745","last_updated":"2026-05-08T07:43:02Z","snapshot_observed_at":"2026-08-05T15:50:53.366518Z","submitted_at":"2026-05-08T07:43:02Z","title":"DocAnnot -- Accelerating the Creation of Key Information Extraction Datasets with GenAI-Powered Auto-annotation","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-02T14:38:35.492534Z"},"links":{"cited_paper":"/paper/2007.00398","citing_paper":"/paper/2607.24745"},"observation_digest":"sha256:741b4e9e27ffd90c6e380d3d7e33f4fdd4ac5c5bdc7e5c46181d93e37bc74266","observation_id":"2774c8d8-9c4f-4ac0-b7c5-6e6d4d973c7e","resolution":{"observed_at":"2026-08-02T14:38:35.492534Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2007.00398","last_updated":"2021-01-05T05:39:39Z","snapshot_observed_at":"2026-07-06T09:34:24.643148Z","submitted_at":"2020-07-01T11:37:40Z","title":"DocVQA: A Dataset for VQA on Document Images","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2007.00398","snapshot_observed_at":"2026-08-01T02:55:23.403485Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.25294","last_updated":"2026-07-28T05:06:43Z","snapshot_observed_at":"2026-08-06T17:57:12.013512Z","submitted_at":"2026-07-28T05:06:43Z","title":"CLBench-V: Evaluating Multimodal Context Learning from Grounding to Knowledge Acquisition","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-01T02:55:23.403485Z"},"links":{"cited_paper":"/paper/2007.00398","citing_paper":"/paper/2607.25294"},"observation_digest":"sha256:f4685d9742e29031bdf27372f723a89df09fb07b7be600b6d80a223581c96a41","observation_id":"86b507e3-44ac-49ba-a61e-a3f591c852f8","resolution":{"observed_at":"2026-08-01T02:55:23.403485Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2007.00398","last_updated":"2021-01-05T05:39:39Z","snapshot_observed_at":"2026-07-06T09:34:24.643148Z","submitted_at":"2020-07-01T11:37:40Z","title":"DocVQA: A Dataset for VQA on Document Images","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2007.00398","snapshot_observed_at":"2026-08-05T04:26:29.325114Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2608.02636","last_updated":"2026-07-31T04:02:51Z","snapshot_observed_at":"2026-08-07T23:09:49.454535Z","submitted_at":"2026-07-31T04:02:51Z","title":"Rethinking Self-Evolving Agent Skills: Feedback Dynamics over Multiple Rounds","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-05T04:26:29.325114Z"},"links":{"cited_paper":"/paper/2007.00398","citing_paper":"/paper/2608.02636"},"observation_digest":"sha256:73f5e861eff5a840a4183142f55e57be4bee37cde36695549ae95483954ee5fd","observation_id":"d129ac94-b538-4ea4-b655-27c8cecd55c1","resolution":{"observed_at":"2026-08-05T04:26:29.325114Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2007.00398/citation-record","integrity":"/paper/2007.00398/integrity","json":"/paper/2007.00398/citation-record.json","paper":"/paper/2007.00398"},"outbound":[],"paper":{"arxiv_id":"2007.00398","last_updated":"2021-01-05T05:39:39Z","latest_version":3,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T09:34:24.643148Z","submitted_at":"2020-07-01T11:37:40Z","title":"DocVQA: A Dataset for VQA on Document Images"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 38 inbound Pith citation observations for arXiv:2007.00398."}