{"as_of":"2026-08-07T08:19:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:182e291b280166288304ce4de175e5854ae7ca50a7b0b6cff6b714b1cd4d94e5","coverage":[{"denominator":36,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":36,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T16:12:03.811310Z","state":"measured"},{"denominator":36,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":36,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2507.15882/citation-record","integrity":"/paper/2507.15882/integrity","json":"/paper/2507.15882/citation-record.json","paper":"/paper/2507.15882"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:12:09.339277Z","title":"Improving language understanding by generative pre-training","venue":null,"work_id":"acabdaa9-2c8c-45a5-870d-6388fab9b21b","year":2018},"citing_paper":{"arxiv_id":"2507.15882","last_updated":"2025-08-04T20:48:37Z","snapshot_observed_at":"2026-08-06T15:56:09.693839Z","submitted_at":"2025-07-18T19:33:15Z","title":"Document Haystack: A Long Context Multimodal Image/Document Understanding Vision LLM Benchmark","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T16:11:59.365713Z"},"links":{"citing_paper":"/paper/2507.15882"},"observation_digest":"sha256:ef379ddf341d822ffd5fba309e2e592387ba253356e0912b94b1ed492bd81062","observation_id":"bdec5a75-e740-4419-9381-c9d9a27ecc92","resolution":{"observed_at":"2026-08-06T16:12:09.403705Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:11:59.401910Z","title":"Palm: Scaling language modeling with pathways","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.15882","last_updated":"2025-08-04T20:48:37Z","snapshot_observed_at":"2026-08-06T15:56:09.693839Z","submitted_at":"2025-07-18T19:33:15Z","title":"Document Haystack: A Long Context Multimodal Image/Document Understanding Vision LLM Benchmark","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T16:11:59.401910Z"},"links":{"citing_paper":"/paper/2507.15882"},"observation_digest":"sha256:ee9cbf1e39d776268cafdeaf150f04e5aab3dc2da0ea38935fd7588feb6a1798","observation_id":"15ff5d91-7beb-4982-a532-37f0e1536a26","resolution":{"observed_at":"2026-08-06T16:11:59.401910Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-06T16:11:59.430437Z","title":"Llama: Open and efficient foundation language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.15882","last_updated":"2025-08-04T20:48:37Z","snapshot_observed_at":"2026-08-06T15:56:09.693839Z","submitted_at":"2025-07-18T19:33:15Z","title":"Document Haystack: A Long Context Multimodal Image/Document Understanding Vision LLM Benchmark","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T16:11:59.430437Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2507.15882"},"observation_digest":"sha256:f9be32a8d39bcca6ab6d66f10a48ed7193a73b63d2b1cd8e06713b18aa452cd9","observation_id":"d7bf8b15-b176-45bb-9de9-0b3e549a0122","resolution":{"observed_at":"2026-08-06T16:11:59.430437Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-02T11:57:18.735747Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-06T16:11:59.455039Z","title":"Llama 2: Open foundation and fine-tuned chat models.arXiv preprint arXiv:2307.09288, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.15882","last_updated":"2025-08-04T20:48:37Z","snapshot_observed_at":"2026-08-06T15:56:09.693839Z","submitted_at":"2025-07-18T19:33:15Z","title":"Document Haystack: A Long Context Multimodal Image/Document Understanding Vision LLM Benchmark","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T16:11:59.455039Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2507.15882"},"observation_digest":"sha256:99b52027461b465ae28ddaa467ee64adc4e5dcb7de7edd980b6fe5a70cfa5634","observation_id":"5ffc3dc8-c6b8-4fe4-822d-8208fa21d627","resolution":{"observed_at":"2026-08-06T16:11:59.455039Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-06T16:11:59.493405Z","title":"Gpt-4 technical report","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.15882","last_updated":"2025-08-04T20:48:37Z","snapshot_observed_at":"2026-08-06T15:56:09.693839Z","submitted_at":"2025-07-18T19:33:15Z","title":"Document Haystack: A Long Context Multimodal Image/Document Understanding Vision LLM Benchmark","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T16:11:59.493405Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2507.15882"},"observation_digest":"sha256:d3bc327edd62adcb233ec114b54ded6235df8154d6a12f6aed7368b2f92ba34c","observation_id":"055f1b8a-079e-4295-a6e0-373421acef82","resolution":{"observed_at":"2026-08-06T16:11:59.493405Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-06T16:11:59.549630Z","title":"Gemini: a family of highly capable multimodal models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.15882","last_updated":"2025-08-04T20:48:37Z","snapshot_observed_at":"2026-08-06T15:56:09.693839Z","submitted_at":"2025-07-18T19:33:15Z","title":"Document Haystack: A Long Context Multimodal Image/Document Understanding Vision LLM Benchmark","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T16:11:59.549630Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2507.15882"},"observation_digest":"sha256:47c9d8c48f59b97bde3f50bc1cefcc9d49b89763dacdb46ddf3ec775b206f9be","observation_id":"afb4f16d-a539-4702-bdc4-b88b33ff2f43","resolution":{"observed_at":"2026-08-06T16:11:59.549630Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-06T16:11:59.677645Z","title":"The llama 3 herd of models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.15882","last_updated":"2025-08-04T20:48:37Z","snapshot_observed_at":"2026-08-06T15:56:09.693839Z","submitted_at":"2025-07-18T19:33:15Z","title":"Document Haystack: A Long Context Multimodal Image/Document Understanding Vision LLM Benchmark","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T16:11:59.677645Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2507.15882"},"observation_digest":"sha256:862a0b374e88ca8041ab36dc48ca503f892bc540891a6dce59cd2cb2fe78a338","observation_id":"746a3555-eb67-43d1-bb57-19cc4bcab9f6","resolution":{"observed_at":"2026-08-06T16:11:59.677645Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:12:09.038247Z","title":"Introducing the next generation of claude: Claude 3 model family","venue":null,"work_id":"ea0d6840-899f-42ae-a149-62f2dd36d515","year":2024},"citing_paper":{"arxiv_id":"2507.15882","last_updated":"2025-08-04T20:48:37Z","snapshot_observed_at":"2026-08-06T15:56:09.693839Z","submitted_at":"2025-07-18T19:33:15Z","title":"Document Haystack: A Long Context Multimodal Image/Document Understanding Vision LLM Benchmark","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T16:11:59.812322Z"},"links":{"citing_paper":"/paper/2507.15882"},"observation_digest":"sha256:7f471f4438816509d24c3032a967bb06f8dc3b330433625dc2bcff0d43297ed2","observation_id":"dd1fa778-bdde-4df4-b501-cc0a34073135","resolution":{"observed_at":"2026-08-06T16:12:09.150309Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:12:08.834376Z","title":"The amazon nova family of models: Technical report and model card","venue":null,"work_id":"540afd58-02e6-49c3-b080-779d4c20a528","year":2024},"citing_paper":{"arxiv_id":"2507.15882","last_updated":"2025-08-04T20:48:37Z","snapshot_observed_at":"2026-08-06T15:56:09.693839Z","submitted_at":"2025-07-18T19:33:15Z","title":"Document Haystack: A Long Context Multimodal Image/Document Understanding Vision LLM Benchmark","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T16:11:59.945149Z"},"links":{"citing_paper":"/paper/2507.15882"},"observation_digest":"sha256:9342bf682b4b3d69b86d620e2ce2e265cc3cc51b6f439447b813b5397ae00cbc","observation_id":"6029eb52-ba7e-4834-b8a9-72a7cc99e65f","resolution":{"observed_at":"2026-08-06T16:12:08.923684Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:12:08.597718Z","title":"Visionllm: Large language model is also an open-ended decoder for vision-centric tasks","venue":null,"work_id":"220db3f7-8be4-4038-b049-e9901db82cfe","year":2024},"citing_paper":{"arxiv_id":"2507.15882","last_updated":"2025-08-04T20:48:37Z","snapshot_observed_at":"2026-08-06T15:56:09.693839Z","submitted_at":"2025-07-18T19:33:15Z","title":"Document Haystack: A Long Context Multimodal Image/Document Understanding Vision LLM Benchmark","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T16:12:00.043620Z"},"links":{"citing_paper":"/paper/2507.15882"},"observation_digest":"sha256:b422cf32bc8fe33aed08a2078283df730e27391918b375007ef964309ead741a","observation_id":"f74e43bc-4295-4db0-a940-1a8ed08f3396","resolution":{"observed_at":"2026-08-06T16:12:08.718134Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.20252","last_updated":"2024-10-26T19:01:06Z","snapshot_observed_at":"2026-07-06T19:40:12.973874Z","submitted_at":"2024-10-26T19:01:06Z","title":"Adaptive Video Understanding Agent: Enhancing efficiency with dynamic frame sampling and feedback-driven reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.20252","snapshot_observed_at":"2026-08-06T16:12:00.161768Z","title":"Adaptive video under- standing agent: Enhancing efficiency with dynamic frame sampling and feedback-driven reasoning","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.15882","last_updated":"2025-08-04T20:48:37Z","snapshot_observed_at":"2026-08-06T15:56:09.693839Z","submitted_at":"2025-07-18T19:33:15Z","title":"Document Haystack: A Long Context Multimodal Image/Document Understanding Vision LLM Benchmark","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T16:12:00.161768Z"},"links":{"cited_paper":"/paper/2410.20252","citing_paper":"/paper/2507.15882"},"observation_digest":"sha256:6ad3fecb74d5c232364a1c178fccedc0482a74628b7048be0eb0570b63c71722","observation_id":"48e270f1-ef79-48ee-9c86-fb54bdc34c40","resolution":{"observed_at":"2026-08-06T16:12:00.161768Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:12:08.312597Z","title":"Zero-resource speech translation and recognition with llms","venue":null,"work_id":"0a5dc98c-3638-4111-8c1e-6b8a33f6842f","year":2025},"citing_paper":{"arxiv_id":"2507.15882","last_updated":"2025-08-04T20:48:37Z","snapshot_observed_at":"2026-08-06T15:56:09.693839Z","submitted_at":"2025-07-18T19:33:15Z","title":"Document Haystack: A Long Context Multimodal Image/Document Understanding Vision LLM Benchmark","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T16:12:00.359797Z"},"links":{"citing_paper":"/paper/2507.15882"},"observation_digest":"sha256:c40376e5a055b4534ab6730d5c8ec8b0750c6d1147181289169c340eecbf0f5a","observation_id":"f70a710e-72fa-49f5-9473-03ad86e397d1","resolution":{"observed_at":"2026-08-06T16:12:08.478594Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:12:08.106207Z","title":"Legal- bench: A collaboratively built benchmark for measuring legal reasoning in large language models","venue":null,"work_id":"ce8df358-bf75-409d-90c9-749199cbda92","year":null},"citing_paper":{"arxiv_id":"2507.15882","last_updated":"2025-08-04T20:48:37Z","snapshot_observed_at":"2026-08-06T15:56:09.693839Z","submitted_at":"2025-07-18T19:33:15Z","title":"Document Haystack: A Long Context Multimodal Image/Document Understanding Vision LLM Benchmark","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T16:12:00.538822Z"},"links":{"citing_paper":"/paper/2507.15882"},"observation_digest":"sha256:20c7c13172289981a63b90e3736301e55a6ccf2820eb4fbe1682348519c2962a","observation_id":"53e7e97b-432f-487e-9f15-55cc23f44d33","resolution":{"observed_at":"2026-08-06T16:12:08.206903Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:12:07.848250Z","title":"Chatlaw: Open-source legal large language model with integrated external knowledge bases","venue":null,"work_id":"3e4a8b83-76da-4c5a-a254-b015ee18ced8","year":2023},"citing_paper":{"arxiv_id":"2507.15882","last_updated":"2025-08-04T20:48:37Z","snapshot_observed_at":"2026-08-06T15:56:09.693839Z","submitted_at":"2025-07-18T19:33:15Z","title":"Document Haystack: A Long Context Multimodal Image/Document Understanding Vision LLM Benchmark","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T16:12:00.658961Z"},"links":{"citing_paper":"/paper/2507.15882"},"observation_digest":"sha256:fb015b1543ab765507e3f2941bd473b348cd42af57eb1d9e01998d0a6dd54164","observation_id":"691f8977-ccee-4b33-91cf-ed45f729094a","resolution":{"observed_at":"2026-08-06T16:12:07.973577Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:12:07.579105Z","title":"A study of generative large language model for medical research and healthcare","venue":null,"work_id":"515eea04-a1c5-4c1d-ab7b-fb73fc272062","year":2023},"citing_paper":{"arxiv_id":"2507.15882","last_updated":"2025-08-04T20:48:37Z","snapshot_observed_at":"2026-08-06T15:56:09.693839Z","submitted_at":"2025-07-18T19:33:15Z","title":"Document Haystack: A Long Context Multimodal Image/Document Understanding Vision LLM Benchmark","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T16:12:00.724539Z"},"links":{"citing_paper":"/paper/2507.15882"},"observation_digest":"sha256:ebc418adbadd08253b4158c70fc18f7c33650e5ec99f6729bb7ae0b9c15b6f55","observation_id":"cfd8f6f6-3ff4-4515-a935-f83a85ea8b94","resolution":{"observed_at":"2026-08-06T16:12:07.652417Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:12:07.378020Z","title":"Large language models in medicine","venue":null,"work_id":"599c75e5-2c47-4ac0-b808-51a1bd5701fc","year":1930},"citing_paper":{"arxiv_id":"2507.15882","last_updated":"2025-08-04T20:48:37Z","snapshot_observed_at":"2026-08-06T15:56:09.693839Z","submitted_at":"2025-07-18T19:33:15Z","title":"Document Haystack: A Long Context Multimodal Image/Document Understanding Vision LLM Benchmark","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T16:12:00.817655Z"},"links":{"citing_paper":"/paper/2507.15882"},"observation_digest":"sha256:66886f8a7aabe7dbf62a1e4cd1462c0775a47396534643d9bad8d8ed00d0993b","observation_id":"5709f843-c7c6-4453-b980-c1b77cd516df","resolution":{"observed_at":"2026-08-06T16:12:07.486410Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:12:07.133641Z","title":"Large language models in finance: A survey","venue":null,"work_id":"cfdfd159-c36b-40b3-88f7-f2c67a66c0eb","year":2023},"citing_paper":{"arxiv_id":"2507.15882","last_updated":"2025-08-04T20:48:37Z","snapshot_observed_at":"2026-08-06T15:56:09.693839Z","submitted_at":"2025-07-18T19:33:15Z","title":"Document Haystack: A Long Context Multimodal Image/Document Understanding Vision LLM Benchmark","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T16:12:00.931170Z"},"links":{"citing_paper":"/paper/2507.15882"},"observation_digest":"sha256:68de03e6ebbcd740f49baaee15d8999aee29603f9dde9bfc9edf195ae1b0ed30","observation_id":"2d67e545-e78e-4c93-93af-b62101901c99","resolution":{"observed_at":"2026-08-06T16:12:07.265020Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.01769","last_updated":"2024-11-21T23:39:12Z","snapshot_observed_at":"2026-08-06T16:39:54.371324Z","submitted_at":"2024-05-02T22:43:02Z","title":"A Survey on Large Language Models for Critical Societal Domains: Finance, Healthcare, and Law","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.01769","snapshot_observed_at":"2026-08-06T16:12:01.088100Z","title":"A survey on large language models for critical societal domains: Finance, healthcare, and law","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15882","last_updated":"2025-08-04T20:48:37Z","snapshot_observed_at":"2026-08-06T15:56:09.693839Z","submitted_at":"2025-07-18T19:33:15Z","title":"Document Haystack: A Long Context Multimodal Image/Document Understanding Vision LLM Benchmark","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T16:12:01.088100Z"},"links":{"cited_paper":"/paper/2405.01769","citing_paper":"/paper/2507.15882"},"observation_digest":"sha256:d20f3b828aee0c427f7bff0c7ed2b10b288fc04a8c44be5337f006402c323f9d","observation_id":"01eccb1b-fd52-43fb-a708-227884a5c295","resolution":{"observed_at":"2026-08-06T16:12:01.088100Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1804.07461","last_updated":"2019-02-22T23:53:34Z","snapshot_observed_at":"2026-07-06T06:34:26.609892Z","submitted_at":"2018-04-20T06:35:04Z","title":"GLUE: A Multi-Task Benchmark and Analysis Platform for Natural Language Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1804.07461","snapshot_observed_at":"2026-08-06T16:12:01.261807Z","title":"Glue: A multi-task benchmark and analysis platform for natural language understanding","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2507.15882","last_updated":"2025-08-04T20:48:37Z","snapshot_observed_at":"2026-08-06T15:56:09.693839Z","submitted_at":"2025-07-18T19:33:15Z","title":"Document Haystack: A Long Context Multimodal Image/Document Understanding Vision LLM Benchmark","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T16:12:01.261807Z"},"links":{"cited_paper":"/paper/1804.07461","citing_paper":"/paper/2507.15882"},"observation_digest":"sha256:625a7f004f9364357386e1ac1cf86b613932b01b8912541ce18884dacb6d5e82","observation_id":"be36be46-14c7-442b-9148-bdf8a426487c","resolution":{"observed_at":"2026-08-06T16:12:01.261807Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:12:06.792063Z","title":"Superglue: A stickier benchmark for general- purpose language understanding systems","venue":null,"work_id":"ddb3274e-bf01-4c8a-875c-90b5a62f2e14","year":2019},"citing_paper":{"arxiv_id":"2507.15882","last_updated":"2025-08-04T20:48:37Z","snapshot_observed_at":"2026-08-06T15:56:09.693839Z","submitted_at":"2025-07-18T19:33:15Z","title":"Document Haystack: A Long Context Multimodal Image/Document Understanding Vision LLM Benchmark","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T16:12:01.394095Z"},"links":{"citing_paper":"/paper/2507.15882"},"observation_digest":"sha256:831a5a9242681c1fb671c85b00ea96b42ad4beda6e10eb21aaaba1efa27cb437","observation_id":"dadc202f-5e84-45c4-a3ff-3e8eceeda7ff","resolution":{"observed_at":"2026-08-06T16:12:06.917425Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:12:06.557262Z","title":"Needle in a haystack-pressure testing llms","venue":null,"work_id":"af0cacbf-36fe-4bea-bd70-5eb29eeb841b","year":2023},"citing_paper":{"arxiv_id":"2507.15882","last_updated":"2025-08-04T20:48:37Z","snapshot_observed_at":"2026-08-06T15:56:09.693839Z","submitted_at":"2025-07-18T19:33:15Z","title":"Document Haystack: A Long Context Multimodal Image/Document Understanding Vision LLM Benchmark","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T16:12:01.566528Z"},"links":{"citing_paper":"/paper/2507.15882"},"observation_digest":"sha256:48c3ba5ae27c1a00507e6a9abf871d2b79021081aab802d57cdc0c2f29d12756","observation_id":"05a1a7e5-78ad-45cf-b5b1-2b7ce80cbff0","resolution":{"observed_at":"2026-08-06T16:12:06.668443Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.14508","last_updated":"2024-06-19T04:00:32Z","snapshot_observed_at":"2026-08-02T11:20:36.216220Z","submitted_at":"2023-08-28T11:53:40Z","title":"LongBench: A Bilingual, Multitask Benchmark for Long Context Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.14508","snapshot_observed_at":"2026-08-06T16:12:01.712774Z","title":"Longbench: A bilingual, multitask benchmark for long context understanding","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.15882","last_updated":"2025-08-04T20:48:37Z","snapshot_observed_at":"2026-08-06T15:56:09.693839Z","submitted_at":"2025-07-18T19:33:15Z","title":"Document Haystack: A Long Context Multimodal Image/Document Understanding Vision LLM Benchmark","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T16:12:01.712774Z"},"links":{"cited_paper":"/paper/2308.14508","citing_paper":"/paper/2507.15882"},"observation_digest":"sha256:ff393a33f62307e0450e1f188495491b853c968bde478fa4ca82d5ac6f314e6b","observation_id":"b703fee8-e780-40c9-a86a-53f5c6c9b6a9","resolution":{"observed_at":"2026-08-06T16:12:01.712774Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:12:06.253672Z","title":"Vqa: Visual question answering","venue":null,"work_id":"de412411-182e-47b3-975c-ee952e13c18c","year":2015},"citing_paper":{"arxiv_id":"2507.15882","last_updated":"2025-08-04T20:48:37Z","snapshot_observed_at":"2026-08-06T15:56:09.693839Z","submitted_at":"2025-07-18T19:33:15Z","title":"Document Haystack: A Long Context Multimodal Image/Document Understanding Vision LLM Benchmark","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T16:12:01.876492Z"},"links":{"citing_paper":"/paper/2507.15882"},"observation_digest":"sha256:7c54c8e51b2114157ee0a7f773a14aa2c822a1513ae589c363e2b57563780c58","observation_id":"e8e3a648-420a-4b26-b5f5-3e32071c7ed3","resolution":{"observed_at":"2026-08-06T16:12:06.382209Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1811.00491","last_updated":"2019-07-21T05:26:36Z","snapshot_observed_at":"2026-08-01T22:56:26.162617Z","submitted_at":"2018-11-01T16:47:44Z","title":"A Corpus for Reasoning About Natural Language Grounded in Photographs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1811.00491","snapshot_observed_at":"2026-08-06T16:12:02.036747Z","title":"A corpus for reasoning about natural language grounded in photographs","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2507.15882","last_updated":"2025-08-04T20:48:37Z","snapshot_observed_at":"2026-08-06T15:56:09.693839Z","submitted_at":"2025-07-18T19:33:15Z","title":"Document Haystack: A Long Context Multimodal Image/Document Understanding Vision LLM Benchmark","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T16:12:02.036747Z"},"links":{"cited_paper":"/paper/1811.00491","citing_paper":"/paper/2507.15882"},"observation_digest":"sha256:8110e009d23997fc478b7d98da95ca543b8d991f9b1f3e4b6661d6c7409a5481","observation_id":"f4e19d24-0934-444b-8fab-2888b4389a5b","resolution":{"observed_at":"2026-08-06T16:12:02.036747Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.18532","last_updated":"2024-05-15T05:43:30Z","snapshot_observed_at":"2026-08-04T01:28:52.686106Z","submitted_at":"2024-04-29T09:19:05Z","title":"MileBench: Benchmarking MLLMs in Long Context","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.18532","snapshot_observed_at":"2026-08-06T16:12:02.171862Z","title":"Milebench: Benchmarking mllms in long context","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.15882","last_updated":"2025-08-04T20:48:37Z","snapshot_observed_at":"2026-08-06T15:56:09.693839Z","submitted_at":"2025-07-18T19:33:15Z","title":"Document Haystack: A Long Context Multimodal Image/Document Understanding Vision LLM Benchmark","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T16:12:02.171862Z"},"links":{"cited_paper":"/paper/2404.18532","citing_paper":"/paper/2507.15882"},"observation_digest":"sha256:c0c4dd221d3cc5a037799764b2cab256e4c537b4739c47d2730c0c332e1a09a1","observation_id":"2c6eae63-2f9c-449f-b856-6080ae010be5","resolution":{"observed_at":"2026-08-06T16:12:02.171862Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.10244","last_updated":"2022-03-19T05:00:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-03-19T05:00:30Z","title":"ChartQA: A Benchmark for Question Answering about Charts with Visual and Logical Reasoning","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.10244","snapshot_observed_at":"2026-08-06T16:12:02.326329Z","title":"Chartqa: A benchmark for question answering about charts with visual and logical reasoning","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.15882","last_updated":"2025-08-04T20:48:37Z","snapshot_observed_at":"2026-08-06T15:56:09.693839Z","submitted_at":"2025-07-18T19:33:15Z","title":"Document Haystack: A Long Context Multimodal Image/Document Understanding Vision LLM Benchmark","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T16:12:02.326329Z"},"links":{"cited_paper":"/paper/2203.10244","citing_paper":"/paper/2507.15882"},"observation_digest":"sha256:b5acc47770d25d00109e8c8a0e83c8ad44d422ab001b96d302b30435b3dc67ca","observation_id":"1e246951-2bed-4f69-9193-0ad1ef4dd56f","resolution":{"observed_at":"2026-08-06T16:12:02.326329Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:12:05.997738Z","title":"Docvqa: A dataset for vqa on document images","venue":null,"work_id":"56bd5a0b-1b5e-4caa-9c6f-d441055e99fb","year":2021},"citing_paper":{"arxiv_id":"2507.15882","last_updated":"2025-08-04T20:48:37Z","snapshot_observed_at":"2026-08-06T15:56:09.693839Z","submitted_at":"2025-07-18T19:33:15Z","title":"Document Haystack: A Long Context Multimodal Image/Document Understanding Vision LLM Benchmark","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T16:12:02.479204Z"},"links":{"citing_paper":"/paper/2507.15882"},"observation_digest":"sha256:9aca9f1f818b7f298a0ce374367dff3b2ba81c83ec026f9a4480ed22a02cf562","observation_id":"76c5ce3d-5e6d-464e-a5d8-2572687aa250","resolution":{"observed_at":"2026-08-06T16:12:06.071480Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:12:05.720107Z","title":"Document understanding dataset and evalua- tion (dude)","venue":null,"work_id":"187848fb-4969-4862-a308-c510a96c0a48","year":2023},"citing_paper":{"arxiv_id":"2507.15882","last_updated":"2025-08-04T20:48:37Z","snapshot_observed_at":"2026-08-06T15:56:09.693839Z","submitted_at":"2025-07-18T19:33:15Z","title":"Document Haystack: A Long Context Multimodal Image/Document Understanding Vision LLM Benchmark","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T16:12:02.641880Z"},"links":{"citing_paper":"/paper/2507.15882"},"observation_digest":"sha256:8e48d00c43a9e46621c21bb3f7c560dc5d24ad1457fbdc65cabe386f1edfed3b","observation_id":"91332362-0274-45a4-a58f-852ee01988dd","resolution":{"observed_at":"2026-08-06T16:12:05.884146Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:12:05.457405Z","title":"Leave no document behind: Benchmarking long-context llms with extended multi-doc qa","venue":null,"work_id":"0bdf7c52-bd60-4dbc-9aa7-8cceb7346219","year":2024},"citing_paper":{"arxiv_id":"2507.15882","last_updated":"2025-08-04T20:48:37Z","snapshot_observed_at":"2026-08-06T15:56:09.693839Z","submitted_at":"2025-07-18T19:33:15Z","title":"Document Haystack: A Long Context Multimodal Image/Document Understanding Vision LLM Benchmark","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T16:12:02.766229Z"},"links":{"citing_paper":"/paper/2507.15882"},"observation_digest":"sha256:cbdfb5f3251c017863c72fb580d2a076d69876f09f6318ec0446ca552349ed0d","observation_id":"eaa94a9d-bd12-43eb-bc08-6fa51c8b11c3","resolution":{"observed_at":"2026-08-06T16:12:05.594366Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:12:05.107142Z","title":"Slidevqa: A dataset for document visual question answering on multiple images","venue":null,"work_id":"c7397c5f-a587-48c7-81b5-500748fcfdf9","year":2023},"citing_paper":{"arxiv_id":"2507.15882","last_updated":"2025-08-04T20:48:37Z","snapshot_observed_at":"2026-08-06T15:56:09.693839Z","submitted_at":"2025-07-18T19:33:15Z","title":"Document Haystack: A Long Context Multimodal Image/Document Understanding Vision LLM Benchmark","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T16:12:02.916848Z"},"links":{"citing_paper":"/paper/2507.15882"},"observation_digest":"sha256:bce4a9d406bfc87d586bf97a13b3e572260619f5ba51c9ec9f36467434557bb7","observation_id":"e8b89465-2f2f-4727-9996-0e454d557092","resolution":{"observed_at":"2026-08-06T16:12:05.270470Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.01523","last_updated":"2024-11-12T04:37:44Z","snapshot_observed_at":"2026-08-06T08:51:10.907429Z","submitted_at":"2024-07-01T17:59:26Z","title":"MMLongBench-Doc: Benchmarking Long-context Document Understanding with Visualizations","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.01523","snapshot_observed_at":"2026-08-06T16:12:02.996244Z","title":"Mmlongbench-doc: Benchmarking long-context document understanding with visualizations","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15882","last_updated":"2025-08-04T20:48:37Z","snapshot_observed_at":"2026-08-06T15:56:09.693839Z","submitted_at":"2025-07-18T19:33:15Z","title":"Document Haystack: A Long Context Multimodal Image/Document Understanding Vision LLM Benchmark","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T16:12:02.996244Z"},"links":{"cited_paper":"/paper/2407.01523","citing_paper":"/paper/2507.15882"},"observation_digest":"sha256:cfb62e2cbe33c044a7fa0f771a06a805c3c57140661a708504309c243bb5e694","observation_id":"d4acb8a5-5d1b-43b1-94a9-f204c153c49a","resolution":{"observed_at":"2026-08-06T16:12:02.996244Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.07230","last_updated":"2024-10-09T07:46:02Z","snapshot_observed_at":"2026-07-06T18:28:51.180903Z","submitted_at":"2024-06-11T13:09:16Z","title":"Needle In A Multimodal Haystack","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.07230","snapshot_observed_at":"2026-08-06T16:12:03.116125Z","title":"Needle in a multimodal haystack","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15882","last_updated":"2025-08-04T20:48:37Z","snapshot_observed_at":"2026-08-06T15:56:09.693839Z","submitted_at":"2025-07-18T19:33:15Z","title":"Document Haystack: A Long Context Multimodal Image/Document Understanding Vision LLM Benchmark","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T16:12:03.116125Z"},"links":{"cited_paper":"/paper/2406.07230","citing_paper":"/paper/2507.15882"},"observation_digest":"sha256:1b254a11a0274689c9616325ccf9bdcab81379f68a622e4a1776c4a241397989","observation_id":"527e72e5-0a7d-4e03-9788-624ac5fab665","resolution":{"observed_at":"2026-08-06T16:12:03.116125Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.06176","last_updated":"2024-11-09T13:30:38Z","snapshot_observed_at":"2026-07-06T19:47:51.117687Z","submitted_at":"2024-11-09T13:30:38Z","title":"M-Longdoc: A Benchmark For Multimodal Super-Long Document Understanding And A Retrieval-Aware Tuning Framework","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.06176","snapshot_observed_at":"2026-08-06T16:12:03.272897Z","title":"M-longdoc: A benchmark for mul- timodal super-long document understanding and a retrieval- aware tuning framework","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.15882","last_updated":"2025-08-04T20:48:37Z","snapshot_observed_at":"2026-08-06T15:56:09.693839Z","submitted_at":"2025-07-18T19:33:15Z","title":"Document Haystack: A Long Context Multimodal Image/Document Understanding Vision LLM Benchmark","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T16:12:03.272897Z"},"links":{"cited_paper":"/paper/2411.06176","citing_paper":"/paper/2507.15882"},"observation_digest":"sha256:e42d3890374fb07199c5ad4f4b0a54ca9ec592e97481f3f6d401c6db4b079597","observation_id":"40d0d1e0-ba4b-4b0a-abe3-874870c8707a","resolution":{"observed_at":"2026-08-06T16:12:03.272897Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07073","last_updated":"2024-10-10T17:59:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-09T17:16:22Z","title":"Pixtral 12B","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.07073","snapshot_observed_at":"2026-08-06T16:12:03.418422Z","title":"Pixtral 12b","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.15882","last_updated":"2025-08-04T20:48:37Z","snapshot_observed_at":"2026-08-06T15:56:09.693839Z","submitted_at":"2025-07-18T19:33:15Z","title":"Document Haystack: A Long Context Multimodal Image/Document Understanding Vision LLM Benchmark","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T16:12:03.418422Z"},"links":{"cited_paper":"/paper/2410.07073","citing_paper":"/paper/2507.15882"},"observation_digest":"sha256:9361a5cfcbfa472144a01be382a8d805ea79287ee23399d7aa85cfc7d0a22136","observation_id":"c64c4e5b-afd7-4f9b-a92e-1e04eed10b3e","resolution":{"observed_at":"2026-08-06T16:12:03.418422Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.14263","last_updated":"2024-09-30T08:17:01Z","snapshot_observed_at":"2026-07-06T16:11:02.763364Z","submitted_at":"2023-08-28T02:38:17Z","title":"Cross-Modal Retrieval: A Systematic Review of Methods and Future Directions","version":3},"cited_work":{"arxiv_id":"2308.14263","doi":null,"metadata_source":"pith","pith_arxiv_id":"2308.14263","snapshot_observed_at":"2026-08-06T16:12:04.020789Z","title":"Cross-Modal Retrieval: A Systematic Review of Methods and Future Directions","venue":"cs.IR","work_id":"ddbce750-0bca-4f0d-ab04-18dce1b8c182","year":2023},"citing_paper":{"arxiv_id":"2507.15882","last_updated":"2025-08-04T20:48:37Z","snapshot_observed_at":"2026-08-06T15:56:09.693839Z","submitted_at":"2025-07-18T19:33:15Z","title":"Document Haystack: A Long Context Multimodal Image/Document Understanding Vision LLM Benchmark","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T16:12:03.638856Z"},"links":{"cited_paper":"/paper/2308.14263","citing_paper":"/paper/2507.15882"},"observation_digest":"sha256:27f06f9cd82b661ba3cf08549b560aa8cc9b9f641870b8a2a3f06d9055c7374f","observation_id":"07604ab5-94a6-416c-8a5b-08e471b4db48","resolution":{"observed_at":"2026-08-06T16:12:04.176645Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T16:12:04.832565Z","title":"Lost in the middle: How language models use long contexts","venue":null,"work_id":"1010e2e0-7381-46a8-84c1-6d78e7064547","year":2024},"citing_paper":{"arxiv_id":"2507.15882","last_updated":"2025-08-04T20:48:37Z","snapshot_observed_at":"2026-08-06T15:56:09.693839Z","submitted_at":"2025-07-18T19:33:15Z","title":"Document Haystack: A Long Context Multimodal Image/Document Understanding Vision LLM Benchmark","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T16:12:03.811310Z"},"links":{"citing_paper":"/paper/2507.15882"},"observation_digest":"sha256:9df1ce14897249b269b0bd83b2b5f86c8cf85e1ab2e8d2e5a9f84648b2953940","observation_id":"8d68d363-5b69-4a39-86db-2c1b14cf346d","resolution":{"observed_at":"2026-08-06T16:12:04.935596Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.15882","last_updated":"2025-08-04T20:48:37Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-06T15:56:09.693839Z","submitted_at":"2025-07-18T19:33:15Z","title":"Document Haystack: A Long Context Multimodal Image/Document Understanding Vision LLM Benchmark"},"reference_resolution":{"displayed":36,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":17,"verified_exact":1,"verified_fuzzy":18},"total_outbound_references":36},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 36 of 36 outbound references and 0 inbound Pith citation observations for arXiv:2507.15882."}