{"as_of":"2026-08-08T07:59:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:9262476bf04f2c36b972b303ecb587f5cd46c598547ed16f2cb5669be2b6ec75","coverage":[{"denominator":192,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T00:24:54.575516Z","state":"measured"},{"denominator":100,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":100,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2608.04424/citation-record","integrity":"/paper/2608.04424/integrity","json":"/paper/2608.04424/citation-record.json","paper":"/paper/2608.04424"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2503.09573","last_updated":"2025-05-17T21:15:02Z","snapshot_observed_at":"2026-08-03T23:49:05.512328Z","submitted_at":"2025-03-12T17:43:40Z","title":"Block Diffusion: Interpolating Between Autoregressive and Diffusion Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.09573","snapshot_observed_at":"2026-08-07T00:24:48.228681Z","title":"Chiu, Zhihan Yang, Zhixuan Qi, Jiaqi Han, Subham Sekhar Sahoo, and Volodymyr Kuleshov","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:48.228681Z"},"links":{"cited_paper":"/paper/2503.09573","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:441cd834cc5b94041a78baef5d3afc1f1449995e24c1adf9e990250fd805e9f3","observation_id":"079bbeaa-c242-4e64-a758-45ab268f868c","resolution":{"observed_at":"2026-08-07T00:24:48.228681Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.12966","last_updated":"2023-10-13T02:41:28Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-24T17:59:17Z","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.12966","snapshot_observed_at":"2026-08-07T00:24:48.289120Z","title":"Qwen-VL: A versatile vision-language model for understanding, localization, text reading, and beyond.arXiv preprint arXiv:2308.12966, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:48.289120Z"},"links":{"cited_paper":"/paper/2308.12966","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:af462f626934441219c75e3cc91680bd25cbfee710082fce4c99362489902842","observation_id":"b87716f9-dcfd-40a0-af94-57f6162a546d","resolution":{"observed_at":"2026-08-07T00:24:48.289120Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.10774","last_updated":"2024-06-14T23:32:32Z","snapshot_observed_at":"2026-07-06T17:17:56.276857Z","submitted_at":"2024-01-19T15:48:40Z","title":"Medusa: Simple LLM Inference Acceleration Framework with Multiple Decoding Heads","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.10774","snapshot_observed_at":"2026-08-07T00:24:48.396919Z","title":"Lee, Deming Chen, and Tri Dao","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:48.396919Z"},"links":{"cited_paper":"/paper/2401.10774","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:14262280e3313b67557bfb0a9d882a6cff9b71a47b1987221ecbfd9294635e0f","observation_id":"939ed157-f89f-4cd0-a570-61b9b2513a3d","resolution":{"observed_at":"2026-08-07T00:24:48.396919Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:48.482296Z","title":"End-to-end object detection with transformers","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:48.482296Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:19a0089d192a86e71330a2206e1288844add6e108a5cf3f217b2f28ec6a0cf48","observation_id":"79d40e6c-d860-48e4-b6e4-e7da9681bfde","resolution":{"observed_at":"2026-08-07T00:24:48.482296Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2511.16719","last_updated":"2026-03-28T16:54:56Z","snapshot_observed_at":"2026-08-07T05:59:39.049027Z","submitted_at":"2025-11-20T18:59:56Z","title":"SAM 3: Segment Anything with Concepts","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2511.16719","snapshot_observed_at":"2026-08-07T00:24:48.560859Z","title":"SAM 3: Segment anything with concepts.arXiv preprint arXiv:2511.16719, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:48.560859Z"},"links":{"cited_paper":"/paper/2511.16719","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:302cf2a9ba17e456a0539db59c7a2ac51940e3bde92d94a3e91d729fa79ce165","observation_id":"6930afa3-d89e-458e-9e14-2c151e7cf105","resolution":{"observed_at":"2026-08-07T00:24:48.560859Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:48.668507Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:48.668507Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:2bbb294773e0426259fa5f8544fcb2e8a3f35af8fc9f81c2eaa7d7a6cc2697ec","observation_id":"89998913-626c-4f99-961b-318e76ce9386","resolution":{"observed_at":"2026-08-07T00:24:48.668507Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1007/978-1-84628-726-8","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T00:03:56.115653Z","title":"Chaudhuri (ed.).Digital Document Processing: Major Directions and Recent Advances","venue":"Advances in pattern recognition","work_id":"f7cfaa77-7766-47e8-85da-ad78bb0ced50","year":2007},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:48.786909Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:3a5f626c782b616cf4a0d6749f96bf81670ed2f12810b6910167a0dc8b80355c","observation_id":"0320ae71-272c-496b-a799-9bf9b37f32e5","resolution":{"observed_at":"2026-08-07T00:24:55.084014Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.15195","last_updated":"2023-07-03T16:08:00Z","snapshot_observed_at":"2026-07-06T15:47:07.545213Z","submitted_at":"2023-06-27T04:31:52Z","title":"Shikra: Unleashing Multimodal LLM's Referential Dialogue Magic","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.15195","snapshot_observed_at":"2026-08-07T00:24:48.866113Z","title":"Shikra: Unleashing multimodal LLM’s referential dialogue magic.arXiv preprint arXiv:2306.15195, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:48.866113Z"},"links":{"cited_paper":"/paper/2306.15195","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:391850c307dff2772f8af57d69a2fe60dfeec734ee89b24f29a28abf094770ec","observation_id":"e98c58e3-c3ae-46d7-a90a-f9730d9cc96c","resolution":{"observed_at":"2026-08-07T00:24:48.866113Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:48.955959Z","title":"Fleet, and Geoffrey Hinton","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:48.955959Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:264637c17626466539fae80266520b8cdd61cb356839391fcf779282da3d0c7f","observation_id":"4c6f4780-02ba-4b86-a7b7-c98724fa0f1b","resolution":{"observed_at":"2026-08-07T00:24:48.955959Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:49.064174Z","title":"Graph-based document structure analysis","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:49.064174Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:ac6db06a4f48d73277ed655f0207a6fa6402279df73136e12de5667d2a155331","observation_id":"551169d8-3bc3-46a9-a6e6-38a9398158c3","resolution":{"observed_at":"2026-08-07T00:24:49.064174Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:49.180375Z","title":"M6doc: A large-scale multi-format, multi-type, multi-layout, multi-language, multi-annotation category dataset for modern document layout analysis","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:49.180375Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:c91bc5b6c8b2da0559855d019bb4873c4cc8b66860dd6be1f92c02e574c314ac","observation_id":"5fae058d-1240-417b-8472-f9b57fc96786","resolution":{"observed_at":"2026-08-07T00:24:49.180375Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:49.262984Z","title":"M 6Doc: A large-scale multi-format, multi-type, multi-layout, multi-language, multi-annotation category dataset for modern document layout analysis","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:49.262984Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:bf7f1b655358b1d8546511fba904fa5b3dd31c7087d34283717052b90db73c89","observation_id":"533c0f0d-e1e9-40c4-a446-fb04145dc1f4","resolution":{"observed_at":"2026-08-07T00:24:49.262984Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:49.348761Z","title":"SDAR: A synergistic diffusion-autoregression paradigm for scalable sequence generation.arXiv preprint arXiv:2510.06303, 2025","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:49.348761Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:c789d0a872f6e0e85217f66abd73d7b33c2b69fa3fdf77c814e1177942e0dbd5","observation_id":"0a506b45-5748-43c7-b9b4-33eb887f46a4","resolution":{"observed_at":"2026-08-07T00:24:49.348761Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.10068","last_updated":"2024-11-15T09:33:13Z","snapshot_observed_at":"2026-07-06T19:50:52.652955Z","submitted_at":"2024-11-15T09:33:13Z","title":"Diachronic Document Dataset for Semantic Layout Analysis","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.10068","snapshot_observed_at":"2026-08-07T00:24:49.478969Z","title":"Diachronic document dataset for semantic layout analysis.arXiv preprint arXiv:2411.10068,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:49.478969Z"},"links":{"cited_paper":"/paper/2411.10068","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:13ff1a0af12fb994f576cf56e6b7dc2b21164a98bf79ba5e63ffcf9f16fb54d0","observation_id":"5dca52eb-1408-4797-bab4-6b0762f268f3","resolution":{"observed_at":"2026-08-07T00:24:49.478969Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:49.678461Z","title":"Molmo and pixmo: Open weights and open data for state-of- the-art vision-language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:49.678461Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:73448c5da322fc11a91274dfedf94691f22fb8d09b24fc88a56071d2b5ecb87d","observation_id":"ee3bbd3d-77a1-44c4-bc5b-5248a3b56d0c","resolution":{"observed_at":"2026-08-07T00:24:49.678461Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2607.06420","last_updated":"2026-07-07T15:48:57Z","snapshot_observed_at":"2026-08-08T07:14:22.144313Z","submitted_at":"2026-07-07T15:48:57Z","title":"HoloCount: A Holistic Visual Counting Benchmark for MLLMs","version":1},"cited_work":{"arxiv_id":"2607.06420","doi":null,"metadata_source":"pith","pith_arxiv_id":"2607.06420","snapshot_observed_at":"2026-08-07T00:24:55.718791Z","title":"HoloCount: A Holistic Visual Counting Benchmark for MLLMs","venue":"cs.CV","work_id":"e6d56d64-7c23-4f18-b100-cad5bfb19d4d","year":2026},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:49.859133Z"},"links":{"cited_paper":"/paper/2607.06420","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:e8512a6557a0b77191763ef20ec06432f4b78baea94a51ff6beabab72801d707","observation_id":"1fc0bc2d-daf6-4fe0-acbe-d41226c90871","resolution":{"observed_at":"2026-08-07T00:24:55.761641Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:49.934247Z","title":null,"venue":null,"work_id":null,"year":2010},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:49.934247Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:78e5bf92b9a9cd1245f158db5e87ad462a42bc74a1901a2d400660bcdee905ec","observation_id":"2fdd163c-3a6d-4f70-830e-8d288f4cb57d","resolution":{"observed_at":"2026-08-07T00:24:49.934247Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2107.08430","last_updated":"2021-08-06T03:22:14Z","snapshot_observed_at":"2026-07-06T11:30:06.143581Z","submitted_at":"2021-07-18T12:55:11Z","title":"YOLOX: Exceeding YOLO Series in 2021","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2107.08430","snapshot_observed_at":"2026-08-07T00:24:50.023097Z","title":"YOLOX: Exceeding YOLO series in 2021.arXiv preprint arXiv:2107.08430, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:50.023097Z"},"links":{"cited_paper":"/paper/2107.08430","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:378b544f39b585f536e19b59000140e979871fc7a54fc90037667ad4d01f7fcd","observation_id":"9d869abe-5c80-455a-aa68-321478e61d8b","resolution":{"observed_at":"2026-08-07T00:24:50.023097Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:50.131818Z","title":"Better & faster large language models via multi-token prediction","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:50.131818Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:4d7c074623ebf742da8095290b1c6877904839b3aff53507f716d9fc224f4875","observation_id":"6f503460-1747-43d4-b41f-fc41ecfe8519","resolution":{"observed_at":"2026-08-07T00:24:50.131818Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:50.211249Z","title":"Morariu, Handong Zhao, Rajiv Jain, Nikolaos Barmpalios, Ani Nenkova, and Tong Sun","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:50.211249Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:2524ff56895405fb5749b742131569590d4e69b1ad5addf0b9a70af1372209b9","observation_id":"d863bb68-d4ae-4d07-ae0b-474357c11d1f","resolution":{"observed_at":"2026-08-07T00:24:50.211249Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:50.284619Z","title":"ADoPD: A large-scale document page decomposition dataset","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:50.284619Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:9346b0b7564094589881ce85ff572e0058cf772a0e58f285fcbdc74fd5f2267b","observation_id":"7bf1e27a-3c7c-41db-a7b8-3332fbfef14c","resolution":{"observed_at":"2026-08-07T00:24:50.284619Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:50.339968Z","title":"Visual programming: Compositional visual reasoning without training","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:50.339968Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:f3f061025420270744847f3302bdb5f8c381aa473731519845e3feb138af17f0","observation_id":"5f9b4020-4b7a-4051-b975-a4c2e83b929a","resolution":{"observed_at":"2026-08-07T00:24:50.339968Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2605.12623","last_updated":"2026-05-21T05:33:02Z","snapshot_observed_at":"2026-08-06T19:38:37.895679Z","submitted_at":"2026-05-12T18:09:38Z","title":"DocAtlas: Multilingual Document Understanding Across 80+ Languages","version":2},"cited_work":{"arxiv_id":"2605.12623","doi":"10.48550/arxiv.2605.12623","metadata_source":"pith","pith_arxiv_id":"2605.12623","snapshot_observed_at":"2026-08-07T06:16:28.064256Z","title":"DocAtlas: Multilingual Document Understanding Across 80+ Languages","venue":"cs.CL","work_id":"c14eacb2-49d5-4cfa-bea7-93a7419c713f","year":2026},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:50.491568Z"},"links":{"cited_paper":"/paper/2605.12623","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:fe4d2a63315d76bd4257f371a0cc23e06f6f15002453c96c2bdfb142512bd944","observation_id":"411e90da-7263-4a79-b97e-a8d49deb4b08","resolution":{"observed_at":"2026-08-07T00:24:55.065353Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:50.590490Z","title":"Drone-based object counting by spatially regularized regional proposal network","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:50.590490Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:a6d7b57953f6cb08e7294a6994df8d41f562591c977c45476b0690160d81dc89","observation_id":"de501bda-bfc5-4cd5-92a2-e5fdad6cf85e","resolution":{"observed_at":"2026-08-07T00:24:50.590490Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:50.674143Z","title":"Layoutlmv3: Pre-training for document ai with unified text and image masking","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:50.674143Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:735c0dfcb92ee1907f4d7d60746c0decb62d86930981c5bac12286396003e871","observation_id":"c32ed399-ba1a-4a3b-a62c-6f3c45762d92","resolution":{"observed_at":"2026-08-07T00:24:50.674143Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:50.779625Z","title":"Composition loss for counting, density map estimation and localization in dense crowds","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:50.779625Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:4619f3dc4ac86fbe12bd197bded27fac2b58d058be33de475c1a2496d93e1df8","observation_id":"45afe1ac-cda1-4449-b81d-fe0632a77905","resolution":{"observed_at":"2026-08-07T00:24:50.779625Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:50.928068Z","title":"OCR-free document understanding transformer","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:50.928068Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:075b5af70ac569b14e85b88ca9e59f2ec97854fbd079b63bc6f763daa324c213","observation_id":"d65af17d-37ba-4d74-a156-8e4f031bbbdd","resolution":{"observed_at":"2026-08-07T00:24:50.928068Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:51.487779Z","title":"Berg, Wan-Yen Lo, Piotr Dollár, and Ross Girshick","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:51.487779Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:e463e9a6006f77d1e15405dc04bc2edc49ed9d6982285057fe79515b111f03fc","observation_id":"896c4cc0-aa0e-4673-972f-6b9426b418c6","resolution":{"observed_at":"2026-08-07T00:24:51.487779Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:51.598760Z","title":"Page segmentation using a convolutional neural network with trainable co-occurrence features","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:51.598760Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:56262430f5ada175a2b0fabe16b901c2a1a0647b9893b172d3ab68b2ddb7cb6f","observation_id":"88d80684-a1b8-4f5b-96b4-a410ad80ced4","resolution":{"observed_at":"2026-08-07T00:24:51.598760Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:51.860465Z","title":"DocBank: A benchmark dataset for document layout analysis","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:51.860465Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:2ede4ac51015766b3f6efa80ebd9be339f9f96e00de6f37125f1f7f122e7bb10","observation_id":"9f90620e-7413-4ee8-9718-02312095bd77","resolution":{"observed_at":"2026-08-07T00:24:51.860465Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:51.949260Z","title":"Morariu, Handong Zhao, Rajiv Jain, Varun Manjunatha, and Hongfu Liu","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:51.949260Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:bd12a3c6430bf9c82623698bb6561df0b40abb26ccaac0b95e6de97d3cdb439b","observation_id":"21620cec-624c-4959-9099-2cf26a4621dc","resolution":{"observed_at":"2026-08-07T00:24:51.949260Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.48550/arxiv.2506.05218","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:55.040056Z","title":"MonkeyOCR: Document parsing with a structure-recognition-relation triplet paradigm.arXiv preprint arXiv:2506.05218, 2025","venue":null,"work_id":"33d54a77-ce6c-4f9d-b85f-606bd9e111f1","year":2025},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:52.027097Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:30f3e3a4a4ffdf010a75b786080a74cb33799484e98d9da6008da51b5f3b2c65","observation_id":"820acda7-e2b4-4fa0-8336-abeae281e46f","resolution":{"observed_at":"2026-08-07T00:24:55.043880Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:52.105752Z","title":"Belongie, James Hays, Pietro Perona, Deva Ramanan, Piotr Dollár, and C","venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:52.105752Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:9c5cdc4548c94a8bad15596da441bc218440c9a0e27a0fff4b063e6cbb118950","observation_id":"77cd14d2-05c4-4874-b781-bdd45a5cdeed","resolution":{"observed_at":"2026-08-07T00:24:52.105752Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:52.183456Z","title":"Visual instruction tuning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:52.183456Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:0aa62966d438ff99907ab00e301c0c4d3db09d6144bd121a9d1f6ad62ce8fe69","observation_id":"f4163009-e78c-427a-bae8-31c41caa1fc4","resolution":{"observed_at":"2026-08-07T00:24:52.183456Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:52.290243Z","title":"Grounding DINO: Marrying DINO with grounded pre-training for open-set object detection","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:52.290243Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:0453b9ef0b5ce6203d4373c7c18c29fc3159aa1981013b8e381b273dbe36ac11","observation_id":"43378107-26ca-4fa4-a554-6d8ab5ef90cc","resolution":{"observed_at":"2026-08-07T00:24:52.290243Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:52.374404Z","title":"Unified-IO 2: Scaling autoregressive multimodal models with vision, language, audio, and action","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:52.374404Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:aa99eaccbbcbac58a06585a599c96b68647fc730b7353d7697d8607e7fd4a65f","observation_id":"be5c1e1f-4a3a-4ab7-a795-8458df0ef21a","resolution":{"observed_at":"2026-08-07T00:24:52.374404Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:52.470880Z","title":"Thinking with visual primitives","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:52.470880Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:525257901084d73a505f0aa8370f4ccd322247e2e75630d0d474174d2ddedcdc","observation_id":"394f9ca7-b8f4-4e03-81a5-0a8b81ea2f44","resolution":{"observed_at":"2026-08-07T00:24:52.470880Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:52.561322Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:52.561322Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:f454666a906f7504415c4050cfb6a4652becc322a9772de263a0190253ad0a97","observation_id":"fac83a38-d9a9-40b1-bf01-69180567a452","resolution":{"observed_at":"2026-08-07T00:24:52.561322Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1609/aaai.v37i2.25282","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.957930Z","title":null,"venue":null,"work_id":"a8c976f8-f5d4-4516-8922-1d6e62836cf6","year":1914},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:52.639233Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:c4de7b061e8100ff9a51683f4c630c36d79d7590d4f65a750ca7b9cd0938db02","observation_id":"b61bccc2-bb04-4df5-9e9e-1886828ffa9a","resolution":{"observed_at":"2026-08-07T00:24:54.960277Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1007/978-3-030-57058-3_16","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T00:03:56.115653Z","title":null,"venue":"Lecture notes in computer science","work_id":"7693515c-da55-4497-8062-ba75f1a411fc","year":2020},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:52.750695Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:78f7a3093fa1914108c8db99c138d39249bef80877d8b1711a9c1e632cb7952a","observation_id":"61f184af-d016-4a8e-b440-6610a124707a","resolution":{"observed_at":"2026-08-07T00:24:54.952861Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:52.824268Z","title":"Iiit-ar-13k: a new dataset for graphical object detection in documents","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:52.824268Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:07cd3a8520dd05d360a87e0b4dcbe6fe97f3f7bc3ac9e5e6a7fea3afb5a1c5db","observation_id":"a9b12337-80b2-4e52-a59f-e6f44547cc9e","resolution":{"observed_at":"2026-08-07T00:24:52.824268Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1007/978-3-032-04614-7_2","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T00:03:56.115653Z","title":"IndicDLP: A foundational dataset for multi-lingual and multi-domain document layout parsing","venue":"Lecture notes in computer science","work_id":"9f58555c-a033-488d-a6c4-cbd3ea1c4249","year":2025},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:52.923043Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:8bf179295063d0efcf0ff1c3aec1b2e76540d709681f7bb95e95ca134c6d110d","observation_id":"b6ae49ec-c851-4144-9fb7-a10bfc31b88a","resolution":{"observed_at":"2026-08-07T00:24:54.944986Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.09992","last_updated":"2025-10-18T15:35:05Z","snapshot_observed_at":"2026-08-04T04:34:22.998376Z","submitted_at":"2025-02-14T08:23:51Z","title":"Large Language Diffusion Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.09992","snapshot_observed_at":"2026-08-07T00:24:53.065685Z","title":"Large language diffusion models.arXiv preprint arXiv:2502.09992, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:53.065685Z"},"links":{"cited_paper":"/paper/2502.09992","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:dcaf87843c8ff1bf2df86466b9594728d2fc76cbf400038f1b68c81af8af865e","observation_id":"41a110d7-a2a8-43ae-aac8-b3c346f3a1cf","resolution":{"observed_at":"2026-08-07T00:24:53.065685Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:53.158010Z","title":"A general approach for multi-oriented text line extraction of handwritten documents.IJDAR, 2012","venue":null,"work_id":null,"year":2012},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:53.158010Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:ff0c66447fddfa177b948cd7ea6302836beb174142a68ba67e68b58d6904a6e2","observation_id":"d7201835-c1b5-4f7e-b8d9-8f9a55eefb83","resolution":{"observed_at":"2026-08-07T00:24:53.158010Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:53.219257Z","title":"Teaching clip to count to ten","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:53.219257Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:a3975b0ce098d7d0866b3a7f3370ae5d402d89391025ebcf397162ae3f4daf35","observation_id":"808f3734-c002-475a-a11d-5faf1cca39c0","resolution":{"observed_at":"2026-08-07T00:24:53.219257Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:53.369671Z","title":"Continuous document layout analysis: Human-in-the-loop AI-based data curation, database, and evaluation in the domain of public affairs.Information Fusion, 108:102398, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:53.369671Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:48a78fda7960aa85967d40850a48c109d0485f0db16e8e4e6e9d15d291e3bb55","observation_id":"58608b4b-dc55-4974-b221-e0d013a8ec6f","resolution":{"observed_at":"2026-08-07T00:24:53.369671Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.14824","last_updated":"2023-07-13T05:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-26T16:32:47Z","title":"Kosmos-2: Grounding Multimodal Large Language Models to the World","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.14824","snapshot_observed_at":"2026-08-07T00:24:53.495972Z","title":"Kosmos-2: Grounding multimodal large language models to the world.arXiv preprint arXiv:2306.14824, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:53.495972Z"},"links":{"cited_paper":"/paper/2306.14824","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:1a0631b56ebee8382d6de7dbfd0261ee30f2acad09f96e238ac196d693716f36","observation_id":"3dc67f54-36c3-4482-8fda-d13e60af90e2","resolution":{"observed_at":"2026-08-07T00:24:53.495972Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:53.740933Z","title":"Nassar, and Peter W","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:53.740933Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:eb69173eb971ff3804fdacaf9001e1c424232023b20db2829f6f58142df03a90","observation_id":"7faada0d-29b7-45e0-b275-7ae08b53f293","resolution":{"observed_at":"2026-08-07T00:24:53.740933Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:53.802744Z","title":"Learning to count everything","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:53.802744Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:736e2a82f807a962cb9e52491503f7c4a7e369a45eeebafd183504a1c5fbfba8","observation_id":"276ccffa-1798-4196-9fd5-131ebe9d0330","resolution":{"observed_at":"2026-08-07T00:24:53.802744Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.00714","last_updated":"2024-10-28T16:37:57Z","snapshot_observed_at":"2026-07-06T18:55:41.459417Z","submitted_at":"2024-08-01T17:00:08Z","title":"SAM 2: Segment Anything in Images and Videos","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.00714","snapshot_observed_at":"2026-08-07T00:24:53.950122Z","title":"SAM 2: Segment anything in images and videos.arXiv preprint arXiv:2408.00714, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:53.950122Z"},"links":{"cited_paper":"/paper/2408.00714","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:43aefa0f0f57693015e6f7f2e36d2b0564714dfa8c85b9aa6768c0d4b1eb6620","observation_id":"ec3cda0e-375b-44a3-b16c-aa0c366f0ed1","resolution":{"observed_at":"2026-08-07T00:24:53.950122Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.060348Z","title":"RF-DETR: Neural architecture search for real-time detection transformers","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.060348Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:2a12986d9bb2118b49da96959b6d113d2df0f6728f169a91257afff267d74777","observation_id":"01ed7edc-18a5-474a-8a50-c52288b54ea3","resolution":{"observed_at":"2026-08-07T00:24:54.060348Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.342985Z","title":"Jhu-crowd++: Large-scale crowd counting dataset and a benchmark method.IEEE transactions on pattern analysis and machine intelligence, 44(5):2594–2609, 2020","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.342985Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:99f22ff310c0a07a7b4bcb8d4aec4898462b8dfcc7e1f429b11daf89a2093dbb","observation_id":"034e7500-57ea-46a7-a51d-dff8000ef3ad","resolution":{"observed_at":"2026-08-07T00:24:54.342985Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.438755Z","title":"An overview of the tesseract OCR engine","venue":null,"work_id":null,"year":2007},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.438755Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:3dd724b3ef895f50e5833034e1592484c7fd48698014bd13de8c9dded4aa57e9","observation_id":"193b4e5c-ae56-4c2f-8b9a-1134a1f869f3","resolution":{"observed_at":"2026-08-07T00:24:54.438755Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.441343Z","title":"ViperGPT: Visual inference via python execution for reasoning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.441343Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:86883f811fb631629fc1227dd8ad53d9368e52caca7cbdbc3ee761432cf0653a","observation_id":"f80cc12e-ffea-4cfc-a0ac-e5ff2cf213ab","resolution":{"observed_at":"2026-08-07T00:24:54.441343Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.06585","last_updated":"2025-09-09T04:46:37Z","snapshot_observed_at":"2026-08-05T23:02:46.819592Z","submitted_at":"2025-08-08T04:23:04Z","title":"CountQA: How Well Do MLLMs Count in the Wild?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.06585","snapshot_observed_at":"2026-08-07T00:24:54.443710Z","title":"Countqa: How well do mllms count in the wild?arXiv preprint arXiv:2508.06585, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.443710Z"},"links":{"cited_paper":"/paper/2508.06585","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:f9c1a9a3167424cae76b1d8fa7135faec64f85795055ae37a413a60f008f4f73","observation_id":"82d90caa-d664-49e6-8f4a-30ba8a1e5e0a","resolution":{"observed_at":"2026-08-07T00:24:54.443710Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.446460Z","title":"Unifying vision, text, and layout for universal document processing","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.446460Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:60500da99be62800e7cb50697837d8d772737e109f55a773d926cc8a667870cc","observation_id":"5576bf1c-6c8e-4710-9832-12241c09914a","resolution":{"observed_at":"2026-08-07T00:24:54.446460Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2606.10651","last_updated":"2026-06-09T09:58:08Z","snapshot_observed_at":"2026-07-06T23:49:51.672382Z","submitted_at":"2026-06-09T09:58:08Z","title":"Kwai Keye-VL-2.0 Technical Report","version":1},"cited_work":{"arxiv_id":"2606.10651","doi":null,"metadata_source":"pith","pith_arxiv_id":"2606.10651","snapshot_observed_at":"2026-08-07T00:24:55.296958Z","title":"Kwai Keye-VL-2.0 Technical Report","venue":"cs.CV","work_id":"57505c5a-161b-4630-94e5-e62535a6ccbf","year":2026},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.448925Z"},"links":{"cited_paper":"/paper/2606.10651","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:45ecb4edc405834446525d05452333f9f1677f0e3be3d6a4f068c6210c64bd39","observation_id":"30ce183f-53e3-4d55-b75e-a072d5c21aeb","resolution":{"observed_at":"2026-08-07T00:24:55.321032Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.12524","last_updated":"2025-02-18T04:20:14Z","snapshot_observed_at":"2026-07-06T20:38:20.545914Z","submitted_at":"2025-02-18T04:20:14Z","title":"YOLOv12: Attention-Centric Real-Time Object Detectors","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.12524","snapshot_observed_at":"2026-08-07T00:24:54.451586Z","title":"YOLOv12: Attention-centric real-time object detectors.arXiv preprint arXiv:2502.12524, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.451586Z"},"links":{"cited_paper":"/paper/2502.12524","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:0a172228e29a0489dc45e066172c8a579026baac836deefdc8563f53ad16a688","observation_id":"819bd54a-be10-4975-be9c-c34d5d50c113","resolution":{"observed_at":"2026-08-07T00:24:54.451586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.454053Z","title":"Tjong Kim Sang and Fien De Meulder","venue":null,"work_id":null,"year":2003},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.454053Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:cc44bb98188412d79b8aa48a7907d561eafbd4a93efc1449721324a4ad3f183f","observation_id":"bfe4410f-5234-4672-8b85-2a1f5e692804","resolution":{"observed_at":"2026-08-07T00:24:54.454053Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.456515Z","title":"SCAN: Semantic document layout analysis for textual and visual retrieval-augmented generation","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.456515Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:7ec5915392c2b215f6fc996832595acd0b023d0bcc640e5bfacc5e49a9a50b99","observation_id":"cb24a83c-1134-40de-b4b8-8d44b04c9ae9","resolution":{"observed_at":"2026-08-07T00:24:54.456515Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.459129Z","title":"Nwpu-crowd: A large-scale benchmark for crowd counting and localization.IEEE transactions on pattern analysis and machine intelligence, 43(6):2141–2149, 2020","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.459129Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:3a231e137fa1b7bbda1dab812becd90b6e4b6cf9d8786649fee85bb6e3ff6154","observation_id":"9698d5c8-eaaa-4402-9ca6-0ebb4acd8e55","resolution":{"observed_at":"2026-08-07T00:24:54.459129Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2605.27365","last_updated":"2026-05-27T02:30:49Z","snapshot_observed_at":"2026-08-06T02:59:09.668446Z","submitted_at":"2026-05-26T17:59:12Z","title":"LocateAnything: Fast and High-Quality Vision-Language Grounding with Parallel Box Decoding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2605.27365","snapshot_observed_at":"2026-08-07T00:24:54.461437Z","title":"Locateanything: Fast and high-quality vision-language grounding with parallel box decoding.arXiv preprint arXiv:2605.27365, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.461437Z"},"links":{"cited_paper":"/paper/2605.27365","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:03d22f5e832fe61835c2c7b9bbd05f2c34c025e83445c9bdd663b90da2a0ab69","observation_id":"4259258c-d8c4-42b8-ae2a-de5784bd6c20","resolution":{"observed_at":"2026-08-07T00:24:54.461437Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.463953Z","title":"Fast-dLLM v2: Efficient block-diffusion LLM.arXiv preprint arXiv:2509.26328, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.463953Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:a1f1f13adb7076bd7da79a3e8e729a799ba4bf39b252107de5d9454ff8df5e28","observation_id":"a690aa5d-53aa-4c4e-8ebe-a2afa9f8a5b6","resolution":{"observed_at":"2026-08-07T00:24:54.463953Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.466365Z","title":"DocGenome: An open large-scale scientific document benchmark for training and testing multi-modal large language models.arXiv preprint arXiv:2406.11633, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.466365Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:2402b06317a5674fa7e7e7404e466f7db211774e8b3ef4a3401710430cdc19ac","observation_id":"d8fe4277-dccc-47ee-9532-5d1ffebcd9a8","resolution":{"observed_at":"2026-08-07T00:24:54.466365Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.469000Z","title":"Florence-2: Advancing a unified representation for a variety of vision tasks","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.469000Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:0f2889e0512b1c43b07f2ae4c0c893d4a39393d66525587c2a92452630ef1196","observation_id":"4cbf61c2-a4cd-4e7c-8e59-d77d14e2081e","resolution":{"observed_at":"2026-08-07T00:24:54.469000Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.11441","last_updated":"2023-11-06T07:39:49Z","snapshot_observed_at":"2026-08-04T03:29:49.409446Z","submitted_at":"2023-10-17T17:51:31Z","title":"Set-of-Mark Prompting Unleashes Extraordinary Visual Grounding in GPT-4V","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.11441","snapshot_observed_at":"2026-08-07T00:24:54.471478Z","title":"Set-of-mark prompting unleashes extraordinary visual grounding in GPT-4V.arXiv preprint arXiv:2310.11441, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.471478Z"},"links":{"cited_paper":"/paper/2310.11441","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:0a63bb752afc16379febed7cfc04118f3148543aef7913dab9fbef1544418371","observation_id":"585d1668-23e4-4831-8573-b8abb2c1dced","resolution":{"observed_at":"2026-08-07T00:24:54.471478Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.11381","last_updated":"2023-03-20T18:31:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-03-20T18:31:47Z","title":"MM-REACT: Prompting ChatGPT for Multimodal Reasoning and Action","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.11381","snapshot_observed_at":"2026-08-07T00:24:54.474041Z","title":"MM-REACT: Prompting chatgpt for multimodal reasoning and action.arXiv preprint arXiv:2303.11381, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.474041Z"},"links":{"cited_paper":"/paper/2303.11381","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:f1f03bcd47854f6bcffdb30b206ed18895232db88423b37a388b7c9bdf137815","observation_id":"fec97968-127e-4f4a-9bcb-371692c94886","resolution":{"observed_at":"2026-08-07T00:24:54.474041Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.15487","last_updated":"2025-08-21T12:09:58Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-08-21T12:09:58Z","title":"Dream 7B: Diffusion Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.15487","snapshot_observed_at":"2026-08-07T00:24:54.476459Z","title":"Dream 7b: Diffusion large language models.arXiv preprint arXiv:2508.15487, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.476459Z"},"links":{"cited_paper":"/paper/2508.15487","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:5fa43c370c7fbc4c3fad0528050a4bc0a64bc60bbd4028e3f0a4d3f96ef4dfc1","observation_id":"712c21a6-6553-49b6-b733-ed53f314de63","resolution":{"observed_at":"2026-08-07T00:24:54.476459Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.479375Z","title":"Ferret: Refer and ground anything anywhere at any granularity","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.479375Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:e87725b8e6a4fe817d446c6573b4be514e6373e5b0702fec4032f3185d88d3c3","observation_id":"8554b5dd-8a74-4452-806c-39494a6757ed","resolution":{"observed_at":"2026-08-07T00:24:54.479375Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.481841Z","title":"Single-image crowd counting via multi- column convolutional neural network","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.481841Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:4c0170bcff5a6a1355e7d6ac79115293e7ef448d01bc251e39704cf2d90caf5d","observation_id":"32fc71cf-f639-4ea2-b614-8097c6899d45","resolution":{"observed_at":"2026-08-07T00:24:54.481841Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.00923","last_updated":"2024-05-20T06:43:48Z","snapshot_observed_at":"2026-07-06T14:47:34.480641Z","submitted_at":"2023-02-02T07:51:19Z","title":"Multimodal Chain-of-Thought Reasoning in Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.00923","snapshot_observed_at":"2026-08-07T00:24:54.484326Z","title":"Multimodal chain-of-thought reasoning in language models.arXiv preprint arXiv:2302.00923, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.484326Z"},"links":{"cited_paper":"/paper/2302.00923","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:35c5f1c3f48bc0e448ffb680d1a6f88a993c113dc684e1726e07459754320f37","observation_id":"9ddfc482-a3c7-4627-a6b0-b25ba5d6ace0","resolution":{"observed_at":"2026-08-07T00:24:54.484326Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.12628","last_updated":"2024-10-16T14:50:47Z","snapshot_observed_at":"2026-08-06T14:52:26.847418Z","submitted_at":"2024-10-16T14:50:47Z","title":"DocLayout-YOLO: Enhancing Document Layout Analysis through Diverse Synthetic Data and Global-to-Local Adaptive Perception","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.12628","snapshot_observed_at":"2026-08-07T00:24:54.487012Z","title":"DocLayout-YOLO: Enhancing document layout analysis through diverse synthetic data and global-to-local adaptive perception.arXiv preprint arXiv:2410.12628,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.487012Z"},"links":{"cited_paper":"/paper/2410.12628","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:5f645cdaab3d645e55eee074c6b49d85409c531fc237e88721aa01e7d9c8f330","observation_id":"76f1fca5-d39d-40bf-9c35-fe62fe81cd9c","resolution":{"observed_at":"2026-08-07T00:24:54.487012Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.495171Z","title":"PubLayNet: Largest dataset ever for document layout analysis","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.495171Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:6ac96a18d2c677bc57f07313a5ba4060f505d9a07866d36403933eac448cc790","observation_id":"747ef930-a335-48d1-9ef0-b22531f0f2e4","resolution":{"observed_at":"2026-08-07T00:24:54.495171Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.12628","last_updated":"2024-10-16T14:50:47Z","snapshot_observed_at":"2026-08-06T14:52:26.847418Z","submitted_at":"2024-10-16T14:50:47Z","title":"DocLayout-YOLO: Enhancing Document Layout Analysis through Diverse Synthetic Data and Global-to-Local Adaptive Perception","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.12628","snapshot_observed_at":"2026-08-07T00:24:54.489763Z","title":"URLhttps://arxiv.org/abs/2410.12628","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.489763Z"},"links":{"cited_paper":"/paper/2410.12628","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:497babcef06caedff65e2d19ce0576b433027a47dcb5d371a5bb0d1eeb9f0c7b","observation_id":"21be752e-a656-41b4-a9a0-f5dcd312670d","resolution":{"observed_at":"2026-08-07T00:24:54.489763Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2606.01774","last_updated":"2026-08-04T08:05:02Z","snapshot_observed_at":"2026-08-07T23:09:28.646647Z","submitted_at":"2026-06-01T06:58:15Z","title":"FLARE: Diffusion for Hybrid Language Model","version":2},"cited_work":{"arxiv_id":"2606.01774","doi":null,"metadata_source":"pith","pith_arxiv_id":"2606.01774","snapshot_observed_at":"2026-08-07T00:24:55.090944Z","title":"FLARE: Diffusion for Hybrid Language Model","venue":"cs.LG","work_id":"ccbe8a64-0fe5-40c3-b9ee-225e10265ac9","year":2026},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.497190Z"},"links":{"cited_paper":"/paper/2606.01774","citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:475e84e2026764b5bee37fd398c2d7a11163c04988d08cd522f6411a2717fd78","observation_id":"c82c4cec-adf2-442f-9f4c-917c7ff3c3f9","resolution":{"observed_at":"2026-08-07T00:24:55.094129Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.500735Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.500735Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:721c039555a1ed5848bb79ef01e76df7bb27bcc9cc2ebb25a8bf1769beb658e2","observation_id":"f7059b72-769a-400f-b7a3-777ceee1a682","resolution":{"observed_at":"2026-08-07T00:24:54.500735Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.502944Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.502944Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:60d4c20a15548b9d4b8f99654e3314311018b65057db7db261ff8737954d3ca8","observation_id":"8ee60e37-69df-4e82-8e5a-05747ca7b8c8","resolution":{"observed_at":"2026-08-07T00:24:54.502944Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.505710Z","title":"icon.A single large, visually striking standalone graphic =prominent pattern","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.505710Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:ffb404a5627a0efdbedeba32beace62ef23d5a080d2fcee78859fc4542941ea9","observation_id":"b9f39d7c-2bdb-4680-a610-450e85432332","resolution":{"observed_at":"2026-08-07T00:24:54.505710Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.508595Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.508595Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:ae5f721df62af60fc82c4295f9537ac9ec53d0910f37ae5c56f6043e71299e0c","observation_id":"17df3003-ab72-472a-a9db-0c63384e8bf6","resolution":{"observed_at":"2026-08-07T00:24:54.508595Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.511061Z","title":"Image labels beside photos =image caption","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.511061Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:1e4c6d8baec7531a0ad175ab84b6055cb0f12ed5002d3c2816df78f5b46851d6","observation_id":"503edb93-da53-495b-a3d4-2d42f43aeaad","resolution":{"observed_at":"2026-08-07T00:24:54.511061Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.513793Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.513793Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:f41a4ede57e4f89141fc25b44bdc9c96d92c8a6ac3e554268850a8aeb3b76e13","observation_id":"82e2858e-b7ac-4263-9c57-333f08db9b7e","resolution":{"observed_at":"2026-08-07T00:24:54.513793Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.516548Z","title":"Plain colored region without border =color block (borderless)","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.516548Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:de0b672c8b6de1e04b177e831bcb232ffd672bcc84f2a71d4b4f690fa85746d8","observation_id":"ca508761-afb6-4de2-bb94-4878bf0ee620","resolution":{"observed_at":"2026-08-07T00:24:54.516548Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.519071Z","title":"Do not split the tag across modes","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.519071Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:81a6f2b6fa7ac5b67801c99f8d52b44cf9c0e95695524a0b16c389ff9588f7cf","observation_id":"38ebcd4d-8e19-4d28-b7cb-4fc0cb202e2a","resolution":{"observed_at":"2026-08-07T00:24:54.519071Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.521947Z","title":"A single-item list must still be separated","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.521947Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:4541f00f1a4c5f0cbf321aefeeb30db43f5f8b9a4ac09295f169b2a247960a72","observation_id":"5bc58f7d-9a72-4852-bf8d-5cc4757c8de8","resolution":{"observed_at":"2026-08-07T00:24:54.521947Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.524518Z","title":"background pattern vs","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.524518Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:1f3e6548a5924275fdf496928dcf37cd1a9220b900f714feb24c2ecf68858105","observation_id":"8b985886-c44d-4bda-bac1-a6238c1db247","resolution":{"observed_at":"2026-08-07T00:24:54.524518Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.527533Z","title":"In newspaper layouts, side-column text adjacent to the main article =note; text at the very bottom =footer","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.527533Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:1d49694999de08a365cd06f9f4a8133338762acc5a2844f5d13e54a3747cc053","observation_id":"535cc545-2d32-426c-8d08-a2589687bb4d","resolution":{"observed_at":"2026-08-07T00:24:54.527533Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.530207Z","title":"Hyperlinks or emphasis embedded in a large body-text block without their own pre-annotation belong to the body-text box","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.530207Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:80ba6331cf395e7365eaf7f95d11f5f1e9360e92e4b0f6da70ffee73c21b71fe","observation_id":"930a991c-7147-4889-b10d-a2062b957a0c","resolution":{"observed_at":"2026-08-07T00:24:54.530207Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.533010Z","title":"image caption; table title.Contextual explanatory text at the top-left or bottom of an image block =note","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.533010Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:0636cee6e63af78d8994b062583d8c741ead9842ca089ced05f82759adb96493","observation_id":"071559b1-af95-4ebc-8c21-30641c358209","resolution":{"observed_at":"2026-08-07T00:24:54.533010Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.535664Z","title":"note.A label directly beside or below a photo =image caption","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.535664Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:84c86ef59a48ec350b5b8a523c26f474b02932f258ccb1d9e73aaf03b85fd373","observation_id":"9e6aefdc-f807-48f2-aeca-48b92022a575","resolution":{"observed_at":"2026-08-07T00:24:54.535664Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.538174Z","title":"All elements above =foreground","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":96,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.538174Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:95a0b47852dcc394758dc5393c2c875976d6bba496c9efe905ea55c735789574","observation_id":"7007ce9d-855c-441b-a333-faaba021672f","resolution":{"observed_at":"2026-08-07T00:24:54.538174Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.541019Z","title":"prominent pattern; legend vs","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.541019Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:0385da17d3e590c7d949622cc72bcef1f920ae05ece7befbec1b4c19aa1d12ca","observation_id":"3c12e59c-14f2-4ab7-9c3a-81f16d34e805","resolution":{"observed_at":"2026-08-07T00:24:54.541019Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.543430Z","title":"legend.Text directly below a photo pointing to it =image caption","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":98,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.543430Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:299051cfd44cb5e80ae1d10a150b0b5db0049b72293862ae944c82926419fa6e","observation_id":"1317472c-6ea6-4b84-ae68-ffce08e77bcc","resolution":{"observed_at":"2026-08-07T00:24:54.543430Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.546004Z","title":"icon (cartoons).Cartoon resembling a recognizable icon-style symbol =icon","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":99,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.546004Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:81a444c976f5640d6a5efd5675f8021b5f2e4a6d65ee25345e2547c6672d2447","observation_id":"adde57ab-746f-41cf-8661-9033fa4c138c","resolution":{"observed_at":"2026-08-07T00:24:54.546004Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.550266Z","title":"background pattern (full-bleed non-solid).If the bottommost layer is a non-solid graphic spanning the full width or height, label itbackground image, notbackground pattern","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":100,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.550266Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:e60c439d3dec1a80b4625c4e35012ff45d82a5c0bd17f3a03327d851256adc27","observation_id":"7ddeea5b-46e3-415f-ac43-68fe1e9ee029","resolution":{"observed_at":"2026-08-07T00:24:54.550266Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.553670Z","title":"There areNnumbered boxes (0..N−1). Group them","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":101,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.553670Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:9eafac71e3e2e9f718127b5ac1ca5f1cf52fde4f0865b84aa7b63e1cba49e1c9","observation_id":"3db1aa9a-c987-4752-a436-e52149246554","resolution":{"observed_at":"2026-08-07T00:24:54.553670Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.559316Z","title":"Below them on the right,[[6]] marks the First in Malaysia badge","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":103,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.559316Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:fb968e4915eb86da40c50a5940429f3de3e944dd6c61b5878b10f05025c56ea4","observation_id":"c8876943-0b61-419d-a001-4362a52b0c31","resolution":{"observed_at":"2026-08-07T00:24:54.559316Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.562342Z","title":"37 Fine-grained Counting [Trigger_Placeholder] How many regions on this page should be counted as Photograph? Original Image Image with Visual Anchors Anchor Thinking Reasoning","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":104,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.562342Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:568c967f49d003f8a1ed57e1a05df3be4a1a7f4a23a5c33601cf63464dccdccc","observation_id":"207ec647-6593-4c59-b01b-4cb8071ff591","resolution":{"observed_at":"2026-08-07T00:24:54.562342Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.567489Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":106,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.567489Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:252ecc138a74c21fe09c21fd0827ac61d79449e9cc1e78eb6e7e91922a4ad3aa","observation_id":"a823d827-ad93-4d7a-acc8-a393b4288557","resolution":{"observed_at":"2026-08-07T00:24:54.567489Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.570262Z","title":"38 Fine-grained Counting [Trigger_Placeholder] Count the Table regions visible on this page","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":107,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.570262Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:14b32b5c5f0e43cf234f079d61887cacf73f05823a0e2e3d510989dc9a9438e8","observation_id":"431c2b81-32ef-45ca-b5ec-52ebf4941ca4","resolution":{"observed_at":"2026-08-07T00:24:54.570262Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T00:24:54.575516Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning","version":1},"reference_index":109,"source":"pdf_text","source_observed_at":"2026-08-07T00:24:54.575516Z"},"links":{"citing_paper":"/paper/2608.04424"},"observation_digest":"sha256:11c1a9b4cdf7004cee8148a031b2ef84243102e191646b52826d6ffd2dea7320","observation_id":"ddc25a1f-2411-43ac-be70-2b6d5cb20636","resolution":{"observed_at":"2026-08-07T00:24:54.575516Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2608.04424","last_updated":"2026-08-05T04:09:05Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-08T07:11:00.607944Z","submitted_at":"2026-08-05T04:09:05Z","title":"Thinking with Anchors: Grounded and Efficient Document Reasoning"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":91,"verified_exact":9,"verified_fuzzy":0},"total_outbound_references":192},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 100 of 192 outbound references and 0 inbound Pith citation observations for arXiv:2608.04424."}