{"as_of":"2026-08-09T18:11:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:b16b9bf332ea1f0fc2cbee5fb6213c68a73a42ac519071449fb710fc4c1ef5ad","coverage":[{"denominator":40,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":40,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-02T06:59:50.917441Z","state":"measured"},{"denominator":40,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":40,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2607.11192/citation-record","integrity":"/paper/2607.11192/integrity","json":"/paper/2607.11192/citation-record.json","paper":"/paper/2607.11192"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T06:59:47.512163Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2607.11192","last_updated":"2026-07-15T07:47:22Z","snapshot_observed_at":"2026-08-02T06:59:46.968270Z","submitted_at":"2026-07-13T07:44:07Z","title":"GDP.pdf: Benchmarking Grounded Multimodal Reasoning over Professional PDF Documents","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-02T06:59:47.512163Z"},"links":{"citing_paper":"/paper/2607.11192"},"observation_digest":"sha256:09583f7e2c2aaee6690601e84fe3971b0f85189b0a49859980d8f8cee0e71b9c","observation_id":"1503f42d-2456-4b72-a690-f76d13f48eaf","resolution":{"observed_at":"2026-08-02T06:59:47.512163Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T06:59:47.618955Z","title":"MEGA-Bench: Scaling multimodal evaluation to over 500 real-world tasks","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.11192","last_updated":"2026-07-15T07:47:22Z","snapshot_observed_at":"2026-08-02T06:59:46.968270Z","submitted_at":"2026-07-13T07:44:07Z","title":"GDP.pdf: Benchmarking Grounded Multimodal Reasoning over Professional PDF Documents","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-02T06:59:47.618955Z"},"links":{"citing_paper":"/paper/2607.11192"},"observation_digest":"sha256:56fecb3f9d19373143ad9d861a3e5a80c834ef1af017b6218cccdd48db284c76","observation_id":"ce926125-01ae-4ea7-bec4-e6c04aaa7034","resolution":{"observed_at":"2026-08-02T06:59:47.618955Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T06:59:47.741062Z","title":"HybridQA: A dataset of multi-hop question answering over tabular and textual data","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2607.11192","last_updated":"2026-07-15T07:47:22Z","snapshot_observed_at":"2026-08-02T06:59:46.968270Z","submitted_at":"2026-07-13T07:44:07Z","title":"GDP.pdf: Benchmarking Grounded Multimodal Reasoning over Professional PDF Documents","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-02T06:59:47.741062Z"},"links":{"citing_paper":"/paper/2607.11192"},"observation_digest":"sha256:3497b5e597b7a11ea088897737834344481c0cf75a088d83ebe5f10ef3841ce7","observation_id":"f42f38fe-2bdb-4d37-bee1-191651d23c39","resolution":{"observed_at":"2026-08-02T06:59:47.741062Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T06:59:47.818325Z","title":"RoDLA: Benchmarking the robustness of document layout analysis models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.11192","last_updated":"2026-07-15T07:47:22Z","snapshot_observed_at":"2026-08-02T06:59:46.968270Z","submitted_at":"2026-07-13T07:44:07Z","title":"GDP.pdf: Benchmarking Grounded Multimodal Reasoning over Professional PDF Documents","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-02T06:59:47.818325Z"},"links":{"citing_paper":"/paper/2607.11192"},"observation_digest":"sha256:c7e05a0a6e52dc796c2c005511db8175962fa073101de236e4e505204e83e650","observation_id":"ae972ef0-121a-4b3a-97da-fa86effe503f","resolution":{"observed_at":"2026-08-02T06:59:47.818325Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T06:59:47.887088Z","title":"FinQA: A dataset of numerical reasoning over financial data","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.11192","last_updated":"2026-07-15T07:47:22Z","snapshot_observed_at":"2026-08-02T06:59:46.968270Z","submitted_at":"2026-07-13T07:44:07Z","title":"GDP.pdf: Benchmarking Grounded Multimodal Reasoning over Professional PDF Documents","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-02T06:59:47.887088Z"},"links":{"citing_paper":"/paper/2607.11192"},"observation_digest":"sha256:ae254e710b054f12df0c305d32cdfa75d32283db08390ce6a2147587326daa10","observation_id":"fa97481b-8b08-42f0-bdd4-ae2a715a85e0","resolution":{"observed_at":"2026-08-02T06:59:47.887088Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T06:59:47.964650Z","title":"M-LongDoc: A benchmark for multimodal super- long document understanding and a retrieval-aware tuning framework","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.11192","last_updated":"2026-07-15T07:47:22Z","snapshot_observed_at":"2026-08-02T06:59:46.968270Z","submitted_at":"2026-07-13T07:44:07Z","title":"GDP.pdf: Benchmarking Grounded Multimodal Reasoning over Professional PDF Documents","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-02T06:59:47.964650Z"},"links":{"citing_paper":"/paper/2607.11192"},"observation_digest":"sha256:dd8d2e27fe4babc227c0e8500c043543fe6b649c1ff5f821e92c1dab57ecaace","observation_id":"1758d779-b0e3-46b8-820f-2fa1926ddd01","resolution":{"observed_at":"2026-08-02T06:59:47.964650Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T06:59:48.084336Z","title":"LongDocURL: a comprehensive multimodal long document benchmark integrating understanding, reason- ing, and locating","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.11192","last_updated":"2026-07-15T07:47:22Z","snapshot_observed_at":"2026-08-02T06:59:46.968270Z","submitted_at":"2026-07-13T07:44:07Z","title":"GDP.pdf: Benchmarking Grounded Multimodal Reasoning over Professional PDF Documents","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-02T06:59:48.084336Z"},"links":{"citing_paper":"/paper/2607.11192"},"observation_digest":"sha256:9ac959bfabacaafea2a47bdf1fc02fd27e614c638292f7c0175e8c88d25111cd","observation_id":"0813694c-b113-495d-8769-8033b83f521c","resolution":{"observed_at":"2026-08-02T06:59:48.084336Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T06:59:48.175604Z","title":"Benchmarking retrieval- augmented multimodal generation for document question answering","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.11192","last_updated":"2026-07-15T07:47:22Z","snapshot_observed_at":"2026-08-02T06:59:46.968270Z","submitted_at":"2026-07-13T07:44:07Z","title":"GDP.pdf: Benchmarking Grounded Multimodal Reasoning over Professional PDF Documents","version":3},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-02T06:59:48.175604Z"},"links":{"citing_paper":"/paper/2607.11192"},"observation_digest":"sha256:409b45a98803205035929bc4f20517c132d1e9e2fd9f09d9fa92d2a235c1e82b","observation_id":"87d17363-7126-457e-a2f9-3888a67aab7e","resolution":{"observed_at":"2026-08-02T06:59:48.175604Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T06:59:48.265432Z","title":"OCRBench v2: An improved benchmark for evaluating large multimodal models on visual text localization and reason- ing","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.11192","last_updated":"2026-07-15T07:47:22Z","snapshot_observed_at":"2026-08-02T06:59:46.968270Z","submitted_at":"2026-07-13T07:44:07Z","title":"GDP.pdf: Benchmarking Grounded Multimodal Reasoning over Professional PDF Documents","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-02T06:59:48.265432Z"},"links":{"citing_paper":"/paper/2607.11192"},"observation_digest":"sha256:d027b6d9bcfaaf94c33a23d9d2eae3308cfe624a20ff3a2ee74aad231158e914","observation_id":"649cf94f-ccb4-4c6c-acac-1efd58691928","resolution":{"observed_at":"2026-08-02T06:59:48.265432Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T06:59:48.354389Z","title":"Ho, Christopher R ´e, Adam Chilton, Aditya Narayana, Alex Chohlas-Wood, Austin Peters, Brandon Waldon, Daniel N","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.11192","last_updated":"2026-07-15T07:47:22Z","snapshot_observed_at":"2026-08-02T06:59:46.968270Z","submitted_at":"2026-07-13T07:44:07Z","title":"GDP.pdf: Benchmarking Grounded Multimodal Reasoning over Professional PDF Documents","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-02T06:59:48.354389Z"},"links":{"citing_paper":"/paper/2607.11192"},"observation_digest":"sha256:80aa61a4788c17b9347b133a332963457693b46ab74cc6633e1204766d672749","observation_id":"bd9df8c1-19e7-4ae0-b385-874eff285fcd","resolution":{"observed_at":"2026-08-02T06:59:48.354389Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.11944","last_updated":"2023-11-20T17:28:02Z","snapshot_observed_at":"2026-07-06T16:50:06.095609Z","submitted_at":"2023-11-20T17:28:02Z","title":"FinanceBench: A New Benchmark for Financial Question Answering","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.11944","snapshot_observed_at":"2026-08-02T06:59:48.470437Z","title":"FinanceBench: A new benchmark for financial question answering.arXiv preprint arXiv:2311.11944, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.11192","last_updated":"2026-07-15T07:47:22Z","snapshot_observed_at":"2026-08-02T06:59:46.968270Z","submitted_at":"2026-07-13T07:44:07Z","title":"GDP.pdf: Benchmarking Grounded Multimodal Reasoning over Professional PDF Documents","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-02T06:59:48.470437Z"},"links":{"cited_paper":"/paper/2311.11944","citing_paper":"/paper/2607.11192"},"observation_digest":"sha256:a8b4da6dd12e1a9d4b4520bc2cff9db76b7134bac28346641a3d430370bb1dc4","observation_id":"de12bad6-9928-4403-a343-baeb8b5b53dc","resolution":{"observed_at":"2026-08-02T06:59:48.470437Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T06:59:48.563708Z","title":"FUNSD: A dataset for form understanding in noisy scanned documents","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2607.11192","last_updated":"2026-07-15T07:47:22Z","snapshot_observed_at":"2026-08-02T06:59:46.968270Z","submitted_at":"2026-07-13T07:44:07Z","title":"GDP.pdf: Benchmarking Grounded Multimodal Reasoning over Professional PDF Documents","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-02T06:59:48.563708Z"},"links":{"citing_paper":"/paper/2607.11192"},"observation_digest":"sha256:4ff1d4430418c1dd05486c074be5575ba1509ca1c9ae3c2d4c0b1bf4807adb84","observation_id":"aea17296-4621-43b4-896b-90b57445ca0e","resolution":{"observed_at":"2026-08-02T06:59:48.563708Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T06:59:48.656576Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.11192","last_updated":"2026-07-15T07:47:22Z","snapshot_observed_at":"2026-08-02T06:59:46.968270Z","submitted_at":"2026-07-13T07:44:07Z","title":"GDP.pdf: Benchmarking Grounded Multimodal Reasoning over Professional PDF Documents","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-02T06:59:48.656576Z"},"links":{"citing_paper":"/paper/2607.11192"},"observation_digest":"sha256:fc11a85d26be2a09ca4307a48ce03255c58f131cfe7c6f2dccec2a2a41139c9a","observation_id":"7a659499-5611-41bc-b05c-349e5c7a6e2a","resolution":{"observed_at":"2026-08-02T06:59:48.656576Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T06:59:48.744101Z","title":"OCRBench: On the hidden mystery of OCR in large multimodal models.Science China Information Sciences, 67(12):220102, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.11192","last_updated":"2026-07-15T07:47:22Z","snapshot_observed_at":"2026-08-02T06:59:46.968270Z","submitted_at":"2026-07-13T07:44:07Z","title":"GDP.pdf: Benchmarking Grounded Multimodal Reasoning over Professional PDF Documents","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-02T06:59:48.744101Z"},"links":{"citing_paper":"/paper/2607.11192"},"observation_digest":"sha256:5dbf1b92569ef31219ee56582dd5d7a0fbf9d2118915372c2cd9882c11638fc2","observation_id":"dcd29ea2-af0e-4f38-8116-a7ec9fcde986","resolution":{"observed_at":"2026-08-02T06:59:48.744101Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T06:59:48.834842Z","title":"MathVista: Evaluating mathemat- ical reasoning of foundation models in visual contexts","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.11192","last_updated":"2026-07-15T07:47:22Z","snapshot_observed_at":"2026-08-02T06:59:46.968270Z","submitted_at":"2026-07-13T07:44:07Z","title":"GDP.pdf: Benchmarking Grounded Multimodal Reasoning over Professional PDF Documents","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-02T06:59:48.834842Z"},"links":{"citing_paper":"/paper/2607.11192"},"observation_digest":"sha256:82db293550259e8964fadf6ad728e339e1d7dabc551de3f369427c399c21b12c","observation_id":"0e8bb9bc-eeca-4549-a62e-76d8d84dc41d","resolution":{"observed_at":"2026-08-02T06:59:48.834842Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T06:59:48.923992Z","title":"MMLongBench-Doc: Bench- marking long-context document understanding with visualiza- tions","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.11192","last_updated":"2026-07-15T07:47:22Z","snapshot_observed_at":"2026-08-02T06:59:46.968270Z","submitted_at":"2026-07-13T07:44:07Z","title":"GDP.pdf: Benchmarking Grounded Multimodal Reasoning over Professional PDF Documents","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-02T06:59:48.923992Z"},"links":{"citing_paper":"/paper/2607.11192"},"observation_digest":"sha256:45fc57f8a53270ae3f093e7f1e62f86f4714ea8da7c5218e93cb540a3087b8ae","observation_id":"f61b70dc-e9b4-4801-bc6f-05934a7a1bdd","resolution":{"observed_at":"2026-08-02T06:59:48.923992Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T06:59:48.992266Z","title":"OK-VQA: A visual question answering benchmark requiring external knowledge","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2607.11192","last_updated":"2026-07-15T07:47:22Z","snapshot_observed_at":"2026-08-02T06:59:46.968270Z","submitted_at":"2026-07-13T07:44:07Z","title":"GDP.pdf: Benchmarking Grounded Multimodal Reasoning over Professional PDF Documents","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-02T06:59:48.992266Z"},"links":{"citing_paper":"/paper/2607.11192"},"observation_digest":"sha256:f7ee9784c1034492760d3b3ca31563e13c5efde05db59a7c2523b8b127754ca8","observation_id":"0af192d1-7a62-4dd6-b9fe-94530adde980","resolution":{"observed_at":"2026-08-02T06:59:48.992266Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T06:59:49.063092Z","title":"ChartQA: A benchmark for question answering about charts with visual and logical reasoning","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.11192","last_updated":"2026-07-15T07:47:22Z","snapshot_observed_at":"2026-08-02T06:59:46.968270Z","submitted_at":"2026-07-13T07:44:07Z","title":"GDP.pdf: Benchmarking Grounded Multimodal Reasoning over Professional PDF Documents","version":3},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-02T06:59:49.063092Z"},"links":{"citing_paper":"/paper/2607.11192"},"observation_digest":"sha256:c754ed4cfaa75cc2b39e54260bdd48dbe1c61b6ad8c2c2d210a5d6c61ba0345d","observation_id":"8b872526-b1bf-4200-bc91-220acd6c05d1","resolution":{"observed_at":"2026-08-02T06:59:49.063092Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T06:59:49.124703Z","title":"ChartQAPro: A more diverse and challenging benchmark for chart question answering","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.11192","last_updated":"2026-07-15T07:47:22Z","snapshot_observed_at":"2026-08-02T06:59:46.968270Z","submitted_at":"2026-07-13T07:44:07Z","title":"GDP.pdf: Benchmarking Grounded Multimodal Reasoning over Professional PDF Documents","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-02T06:59:49.124703Z"},"links":{"citing_paper":"/paper/2607.11192"},"observation_digest":"sha256:1bf9de7e478f96598e7f72c2334930440509f69b25f1c1b06c63c298b27f5bb2","observation_id":"f486bcc6-a4bd-4e26-b4e0-0a7a3b1624b3","resolution":{"observed_at":"2026-08-02T06:59:49.124703Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T06:59:49.229554Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.11192","last_updated":"2026-07-15T07:47:22Z","snapshot_observed_at":"2026-08-02T06:59:46.968270Z","submitted_at":"2026-07-13T07:44:07Z","title":"GDP.pdf: Benchmarking Grounded Multimodal Reasoning over Professional PDF Documents","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-02T06:59:49.229554Z"},"links":{"citing_paper":"/paper/2607.11192"},"observation_digest":"sha256:20a3b5b393dec0be646a13ef694abf92bccdc74937010eb35d113f663d519363","observation_id":"84d2dff2-ebdc-4450-8bc6-b2df960b5754","resolution":{"observed_at":"2026-08-02T06:59:49.229554Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T06:59:49.315744Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.11192","last_updated":"2026-07-15T07:47:22Z","snapshot_observed_at":"2026-08-02T06:59:46.968270Z","submitted_at":"2026-07-13T07:44:07Z","title":"GDP.pdf: Benchmarking Grounded Multimodal Reasoning over Professional PDF Documents","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-02T06:59:49.315744Z"},"links":{"citing_paper":"/paper/2607.11192"},"observation_digest":"sha256:f8270727755f57b9cfd445719bb3c4995e2874af148e5043b4211cd62f0294a7","observation_id":"3d3e52b5-44f5-44ea-a4d3-1998f019f6d4","resolution":{"observed_at":"2026-08-02T06:59:49.315744Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T06:59:49.400835Z","title":"Khapra, and Pratyush Kumar","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2607.11192","last_updated":"2026-07-15T07:47:22Z","snapshot_observed_at":"2026-08-02T06:59:46.968270Z","submitted_at":"2026-07-13T07:44:07Z","title":"GDP.pdf: Benchmarking Grounded Multimodal Reasoning over Professional PDF Documents","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-02T06:59:49.400835Z"},"links":{"citing_paper":"/paper/2607.11192"},"observation_digest":"sha256:3dff16552b5a7ba512005ea3f0a0b2947fce74d10cc9fe5eaf4e3ed9795f1b43","observation_id":"92254d90-4566-446c-9797-8bb1fda97fae","resolution":{"observed_at":"2026-08-02T06:59:49.400835Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T06:59:49.493409Z","title":"OmniDocBench: Benchmarking di- verse PDF document parsing with comprehensive annotations","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.11192","last_updated":"2026-07-15T07:47:22Z","snapshot_observed_at":"2026-08-02T06:59:46.968270Z","submitted_at":"2026-07-13T07:44:07Z","title":"GDP.pdf: Benchmarking Grounded Multimodal Reasoning over Professional PDF Documents","version":3},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-02T06:59:49.493409Z"},"links":{"citing_paper":"/paper/2607.11192"},"observation_digest":"sha256:ec73f9b73fb8fa8f8db30d323ad892e83f173c451156fba48e2ad5e0759ff721","observation_id":"a7d6c6ae-7b8a-4a9c-a0b9-07b8405d0de1","resolution":{"observed_at":"2026-08-02T06:59:49.493409Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T06:59:49.593004Z","title":"CORD: A consolidated receipt dataset for post-OCR parsing","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2607.11192","last_updated":"2026-07-15T07:47:22Z","snapshot_observed_at":"2026-08-02T06:59:46.968270Z","submitted_at":"2026-07-13T07:44:07Z","title":"GDP.pdf: Benchmarking Grounded Multimodal Reasoning over Professional PDF Documents","version":3},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-02T06:59:49.593004Z"},"links":{"citing_paper":"/paper/2607.11192"},"observation_digest":"sha256:1ce1936c263138dfaece29a3e655e81c513fe9e7c486d04886b23c7e32d4e7c2","observation_id":"f42ddebf-350c-41b7-bf56-8a982b89b375","resolution":{"observed_at":"2026-08-02T06:59:49.593004Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03848","last_updated":"2025-04-22T14:11:50Z","snapshot_observed_at":"2026-08-06T01:28:29.570476Z","submitted_at":"2024-02-06T09:50:08Z","title":"ANLS* -- A Universal Document Processing Metric for Generative Large Language Models","version":10},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.03848","snapshot_observed_at":"2026-08-02T06:59:49.672205Z","title":"ANLS* – a universal document processing metric for generative large language models.arXiv preprint arXiv:2402.03848, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.11192","last_updated":"2026-07-15T07:47:22Z","snapshot_observed_at":"2026-08-02T06:59:46.968270Z","submitted_at":"2026-07-13T07:44:07Z","title":"GDP.pdf: Benchmarking Grounded Multimodal Reasoning over Professional PDF Documents","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-02T06:59:49.672205Z"},"links":{"cited_paper":"/paper/2402.03848","citing_paper":"/paper/2607.11192"},"observation_digest":"sha256:92f8a82d9d14ba53df5a6e28e52091bf33ce062ccfecfd73e3ebabbdede426ca","observation_id":"24b8ed1b-db6d-42ad-a10d-5b3f8fb1cedd","resolution":{"observed_at":"2026-08-02T06:59:49.672205Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T06:59:49.761158Z","title":"Nassar, and Peter W","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.11192","last_updated":"2026-07-15T07:47:22Z","snapshot_observed_at":"2026-08-02T06:59:46.968270Z","submitted_at":"2026-07-13T07:44:07Z","title":"GDP.pdf: Benchmarking Grounded Multimodal Reasoning over Professional PDF Documents","version":3},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-02T06:59:49.761158Z"},"links":{"citing_paper":"/paper/2607.11192"},"observation_digest":"sha256:58cb56d84f9eca3e8ab11c33697f19aa83f13ea7d2218124417f394887971778","observation_id":"80ecc4cf-06e7-4946-8282-1f8d7d04cfe9","resolution":{"observed_at":"2026-08-02T06:59:49.761158Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T06:59:49.823769Z","title":"Rossi, and Franck Dernoncourt","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.11192","last_updated":"2026-07-15T07:47:22Z","snapshot_observed_at":"2026-08-02T06:59:46.968270Z","submitted_at":"2026-07-13T07:44:07Z","title":"GDP.pdf: Benchmarking Grounded Multimodal Reasoning over Professional PDF Documents","version":3},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-02T06:59:49.823769Z"},"links":{"citing_paper":"/paper/2607.11192"},"observation_digest":"sha256:2406dba538c0e4166cab79e6fdaf1757c2f8abd8f3166e73461a43646af844dc","observation_id":"8c11d521-ab77-4be5-ad5f-bad25e686ef9","resolution":{"observed_at":"2026-08-02T06:59:49.823769Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T06:59:49.901797Z","title":"A-OKVQA: A benchmark for visual question answering using world knowl- edge","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.11192","last_updated":"2026-07-15T07:47:22Z","snapshot_observed_at":"2026-08-02T06:59:46.968270Z","submitted_at":"2026-07-13T07:44:07Z","title":"GDP.pdf: Benchmarking Grounded Multimodal Reasoning over Professional PDF Documents","version":3},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-02T06:59:49.901797Z"},"links":{"citing_paper":"/paper/2607.11192"},"observation_digest":"sha256:29109b68cb7cfb1483729e03aa7dec789d3867a200301e75b1c019c970175c79","observation_id":"ede9e7f1-8e60-4b31-9358-c1f87da616a1","resolution":{"observed_at":"2026-08-02T06:59:49.901797Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T06:59:50.020674Z","title":"DocILE benchmark for document information localization and extraction","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.11192","last_updated":"2026-07-15T07:47:22Z","snapshot_observed_at":"2026-08-02T06:59:46.968270Z","submitted_at":"2026-07-13T07:44:07Z","title":"GDP.pdf: Benchmarking Grounded Multimodal Reasoning over Professional PDF Documents","version":3},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-02T06:59:50.020674Z"},"links":{"citing_paper":"/paper/2607.11192"},"observation_digest":"sha256:6034f578c1547f181ed2c5ac0c6dc8cfb020dfcd38b957c6daacfedc5b89beee","observation_id":"79f09951-8e71-4775-b55f-9c5e2e53a30f","resolution":{"observed_at":"2026-08-02T06:59:50.020674Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T06:59:50.104411Z","title":"Hi- erarchical multimodal transformers for Multipage DocVQA","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.11192","last_updated":"2026-07-15T07:47:22Z","snapshot_observed_at":"2026-08-02T06:59:46.968270Z","submitted_at":"2026-07-13T07:44:07Z","title":"GDP.pdf: Benchmarking Grounded Multimodal Reasoning over Professional PDF Documents","version":3},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-02T06:59:50.104411Z"},"links":{"citing_paper":"/paper/2607.11192"},"observation_digest":"sha256:bbf8db36f1cdfe152e04f19b63de3dd8eaeb033dc0a89e3b08267613161a491d","observation_id":"92e3c209-609e-49f7-a35d-f78ef6f23148","resolution":{"observed_at":"2026-08-02T06:59:50.104411Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T06:59:50.194909Z","title":"Document understanding dataset and evaluation (DUDE)","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.11192","last_updated":"2026-07-15T07:47:22Z","snapshot_observed_at":"2026-08-02T06:59:46.968270Z","submitted_at":"2026-07-13T07:44:07Z","title":"GDP.pdf: Benchmarking Grounded Multimodal Reasoning over Professional PDF Documents","version":3},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-02T06:59:50.194909Z"},"links":{"citing_paper":"/paper/2607.11192"},"observation_digest":"sha256:874f87f7bd976204e3e6f6066107116104994ce6b4afd2f1c0bb642e7c6ff2fe","observation_id":"c82a8dbe-274e-4496-897b-efe72a965661","resolution":{"observed_at":"2026-08-02T06:59:50.194909Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T06:59:50.290926Z","title":"CharXiv: Charting gaps in realistic chart understand- ing in multimodal LLMs","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.11192","last_updated":"2026-07-15T07:47:22Z","snapshot_observed_at":"2026-08-02T06:59:46.968270Z","submitted_at":"2026-07-13T07:44:07Z","title":"GDP.pdf: Benchmarking Grounded Multimodal Reasoning over Professional PDF Documents","version":3},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-02T06:59:50.290926Z"},"links":{"citing_paper":"/paper/2607.11192"},"observation_digest":"sha256:3bdfb2b0e01843ac7feece46e928c769ac30ab1111f2ca9b63c89e2f0d4fa0d7","observation_id":"8bebf2e7-3eaf-415c-ab46-f8c169c7e410","resolution":{"observed_at":"2026-08-02T06:59:50.290926Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T06:59:50.379963Z","title":"MMMU: A massive multi-discipline multimodal understanding and reasoning benchmark for expert AGI","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.11192","last_updated":"2026-07-15T07:47:22Z","snapshot_observed_at":"2026-08-02T06:59:46.968270Z","submitted_at":"2026-07-13T07:44:07Z","title":"GDP.pdf: Benchmarking Grounded Multimodal Reasoning over Professional PDF Documents","version":3},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-02T06:59:50.379963Z"},"links":{"citing_paper":"/paper/2607.11192"},"observation_digest":"sha256:91ad6ba936182b9f58540df783a5c3b773944d68f04bba8fb4f34d98f9715c3a","observation_id":"d60f0562-17b8-4307-bc4e-7ce972cc611f","resolution":{"observed_at":"2026-08-02T06:59:50.379963Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T06:59:50.498953Z","title":"MMMU-Pro: A more robust multi-discipline multimodal understanding benchmark","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.11192","last_updated":"2026-07-15T07:47:22Z","snapshot_observed_at":"2026-08-02T06:59:46.968270Z","submitted_at":"2026-07-13T07:44:07Z","title":"GDP.pdf: Benchmarking Grounded Multimodal Reasoning over Professional PDF Documents","version":3},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-02T06:59:50.498953Z"},"links":{"citing_paper":"/paper/2607.11192"},"observation_digest":"sha256:079788ca6ca5753e5bf05af07e8418dd455c974d35419ffeb13895032be46fc6","observation_id":"040927e4-e03f-4c22-8e85-47ea0024c31b","resolution":{"observed_at":"2026-08-02T06:59:50.498953Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T06:59:50.558777Z","title":"Pub- LayNet: Largest dataset ever for document layout analysis","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2607.11192","last_updated":"2026-07-15T07:47:22Z","snapshot_observed_at":"2026-08-02T06:59:46.968270Z","submitted_at":"2026-07-13T07:44:07Z","title":"GDP.pdf: Benchmarking Grounded Multimodal Reasoning over Professional PDF Documents","version":3},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-02T06:59:50.558777Z"},"links":{"citing_paper":"/paper/2607.11192"},"observation_digest":"sha256:543593cfa6526b55aba92d042c3c9acd9350703584920d1effcb3df2fcf957d2","observation_id":"22fbf20f-1d95-48c8-accb-db24db9354b6","resolution":{"observed_at":"2026-08-02T06:59:50.558777Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2603.04205","last_updated":"2026-06-22T03:43:39Z","snapshot_observed_at":"2026-08-02T18:54:54.324790Z","submitted_at":"2026-03-04T15:49:06Z","title":"Real5-OmniDocBench: A Full-Scale Physical Reconstruction Benchmark for Robust Document Parsing in the Wild","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2603.04205","snapshot_observed_at":"2026-08-02T06:59:50.615006Z","title":"Real5-OmniDocBench: A full-scale physical reconstruction benchmark for robust docu- ment parsing in the wild.arXiv preprint arXiv:2603.04205,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.11192","last_updated":"2026-07-15T07:47:22Z","snapshot_observed_at":"2026-08-02T06:59:46.968270Z","submitted_at":"2026-07-13T07:44:07Z","title":"GDP.pdf: Benchmarking Grounded Multimodal Reasoning over Professional PDF Documents","version":3},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-02T06:59:50.615006Z"},"links":{"cited_paper":"/paper/2603.04205","citing_paper":"/paper/2607.11192"},"observation_digest":"sha256:0b54736eff93d93c93098650ff477bf66bae57f31e36545ea2247850b2e298c3","observation_id":"5918a48f-22b1-4db9-a739-69fd7559c5d8","resolution":{"observed_at":"2026-08-02T06:59:50.615006Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T06:59:50.686542Z","title":"TAT-QA: A question answering benchmark on a hybrid of tabular and textual content in finance","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.11192","last_updated":"2026-07-15T07:47:22Z","snapshot_observed_at":"2026-08-02T06:59:46.968270Z","submitted_at":"2026-07-13T07:44:07Z","title":"GDP.pdf: Benchmarking Grounded Multimodal Reasoning over Professional PDF Documents","version":3},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-02T06:59:50.686542Z"},"links":{"citing_paper":"/paper/2607.11192"},"observation_digest":"sha256:a4eb729a60e22acc05d47651cd34e03d4c90ad927dd2c893a8978b54e4f8b20a","observation_id":"d0f1e0a6-8b21-41fc-8b51-2c919fd545f9","resolution":{"observed_at":"2026-08-02T06:59:50.686542Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T06:59:50.761784Z","title":"Towards complex document understanding by discrete reasoning","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.11192","last_updated":"2026-07-15T07:47:22Z","snapshot_observed_at":"2026-08-02T06:59:46.968270Z","submitted_at":"2026-07-13T07:44:07Z","title":"GDP.pdf: Benchmarking Grounded Multimodal Reasoning over Professional PDF Documents","version":3},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-02T06:59:50.761784Z"},"links":{"citing_paper":"/paper/2607.11192"},"observation_digest":"sha256:7ef5756aad278e19816cfcd29b2cc17448301ce633e340901d403075c60b0189","observation_id":"64126c30-e86a-46ef-8b1d-4e2f06a934b2","resolution":{"observed_at":"2026-08-02T06:59:50.761784Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-02T06:59:50.862993Z","title":"MMDocBench: Benchmarking large vision-language models for fine-grained visual document understanding and grounding","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.11192","last_updated":"2026-07-15T07:47:22Z","snapshot_observed_at":"2026-08-02T06:59:46.968270Z","submitted_at":"2026-07-13T07:44:07Z","title":"GDP.pdf: Benchmarking Grounded Multimodal Reasoning over Professional PDF Documents","version":3},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-02T06:59:50.862993Z"},"links":{"citing_paper":"/paper/2607.11192"},"observation_digest":"sha256:737756eb56fca5d2fe1c3e06ec688356af11f0418f3b7c36805ae83f7bcb966b","observation_id":"b6e7b5f7-4d36-451c-a7fc-3ee6d0384bc1","resolution":{"observed_at":"2026-08-02T06:59:50.862993Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.10701","last_updated":"2024-07-15T13:17:42Z","snapshot_observed_at":"2026-07-06T18:46:26.900364Z","submitted_at":"2024-07-15T13:17:42Z","title":"DOCBENCH: A Benchmark for Evaluating LLM-based Document Reading Systems","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.10701","snapshot_observed_at":"2026-08-02T06:59:50.917441Z","title":"DOCBENCH: A benchmark for evaluating LLM-based doc- ument reading systems.arXiv preprint arXiv:2407.10701,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.11192","last_updated":"2026-07-15T07:47:22Z","snapshot_observed_at":"2026-08-02T06:59:46.968270Z","submitted_at":"2026-07-13T07:44:07Z","title":"GDP.pdf: Benchmarking Grounded Multimodal Reasoning over Professional PDF Documents","version":3},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-02T06:59:50.917441Z"},"links":{"cited_paper":"/paper/2407.10701","citing_paper":"/paper/2607.11192"},"observation_digest":"sha256:8a47f5269db13175d8bb014c195ace383bbd5e2462aebf92d0b1e0709e84a1fd","observation_id":"bfd442fb-a8db-4fb9-8d76-8b659cd90987","resolution":{"observed_at":"2026-08-02T06:59:50.917441Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2607.11192","last_updated":"2026-07-15T07:47:22Z","latest_version":3,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-02T06:59:46.968270Z","submitted_at":"2026-07-13T07:44:07Z","title":"GDP.pdf: Benchmarking Grounded Multimodal Reasoning over Professional PDF Documents"},"reference_resolution":{"displayed":40,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":40,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":40},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 40 of 40 outbound references and 0 inbound Pith citation observations for arXiv:2607.11192."}