{"as_of":"2026-08-06T15:25:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:6776fd70ffb36e5cc82bc4d0ff629418d5a8fe50fc37a63820d8c4c328ea21ae","coverage":[{"denominator":80,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":80,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-23T19:39:35.147671Z","state":"measured"},{"denominator":85,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":85,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":5,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":5,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-11T16:02:00.920066Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-05-23T19:15:47.037493Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"cited_work":{"arxiv_id":"2410.05970","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.05970","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","venue":"cs.CV","work_id":"d941acb4-8b03-40f4-93cc-e11fdeca8e67","year":2024},"citing_paper":{"arxiv_id":"2410.21169","last_updated":"2026-04-04T17:04:02Z","snapshot_observed_at":"2026-07-06T19:40:52.844113Z","submitted_at":"2024-10-28T16:11:35Z","title":"Document Parsing Unveiled: Techniques, Challenges, and Prospects for Structured Information Extraction","version":5},"reference_index":267,"source":"pdf_text","source_observed_at":"2026-05-23T19:15:21.695801Z"},"links":{"cited_paper":"/paper/2410.05970","citing_paper":"/paper/2410.21169"},"observation_digest":"sha256:3f300e4d4534ff07557ba1d3961a4175dc9bb666d629052a5b28f217a27c6965","observation_id":"67f46f78-935c-4fff-a7d4-fdb834cb75da","resolution":{"observed_at":"2026-05-23T19:15:47.039772Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"cited_work":{"arxiv_id":"2410.05970","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.05970","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","venue":"cs.CV","work_id":"d941acb4-8b03-40f4-93cc-e11fdeca8e67","year":2024},"citing_paper":{"arxiv_id":"2507.09861","last_updated":"2026-04-21T13:31:05Z","snapshot_observed_at":"2026-07-06T21:56:35.665677Z","submitted_at":"2025-07-14T02:10:31Z","title":"A Survey on MLLM-based Visually Rich Document Understanding: Methods, Challenges, and Emerging Trends","version":2},"reference_index":62,"source":"arxiv_source","source_observed_at":"2026-05-19T04:38:49.512293Z"},"links":{"cited_paper":"/paper/2410.05970","citing_paper":"/paper/2507.09861"},"observation_digest":"sha256:d669577cccea60dfdf93af8840215a48093c9be2875631cd9f2023ce71fd9656","observation_id":"856f71ad-cd05-454d-bcee-156a91a6b51c","resolution":{"observed_at":"2026-05-19T04:42:04.371481Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"cited_work":{"arxiv_id":"2410.05970","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.05970","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","venue":"cs.CV","work_id":"d941acb4-8b03-40f4-93cc-e11fdeca8e67","year":2024},"citing_paper":{"arxiv_id":"2604.12812","last_updated":"2026-05-11T03:47:15Z","snapshot_observed_at":"2026-07-06T23:00:56.449281Z","submitted_at":"2026-04-14T14:39:26Z","title":"DocSeeker: Structured Visual Reasoning with Evidence Grounding for Long Document Understanding","version":4},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-10T16:16:58.889065Z"},"links":{"cited_paper":"/paper/2410.05970","citing_paper":"/paper/2604.12812"},"observation_digest":"sha256:c29e82b2dafd1e6d6f679a78a9ebfb3fdf37b4574f7d0b529da8084ff7a0ac5a","observation_id":"e51baeb6-2042-4ee9-a0e3-a0293a368abd","resolution":{"observed_at":"2026-05-11T09:05:58.504487Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"cited_work":{"arxiv_id":"2410.05970","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.05970","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","venue":"cs.CV","work_id":"d941acb4-8b03-40f4-93cc-e11fdeca8e67","year":2024},"citing_paper":{"arxiv_id":"2604.12812","last_updated":"2026-05-11T03:47:15Z","snapshot_observed_at":"2026-07-06T23:00:56.449281Z","submitted_at":"2026-04-14T14:39:26Z","title":"DocSeeker: Structured Visual Reasoning with Evidence Grounding for Long Document Understanding","version":5},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-12T04:17:55.318813Z"},"links":{"cited_paper":"/paper/2410.05970","citing_paper":"/paper/2604.12812"},"observation_digest":"sha256:f13189509ce5c658f5904af6cc2a708435f7d56dde0b51771b6f9e952252fe74","observation_id":"9ec09e0c-fea9-4777-8354-bbaacdb71405","resolution":{"observed_at":"2026-05-12T06:26:24.394356Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.05970","snapshot_observed_at":"2026-07-11T16:02:00.920066Z","title":"arXiv preprint arXiv:2410.05970 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.04636","last_updated":"2026-07-06T03:39:08Z","snapshot_observed_at":"2026-08-06T12:41:38.313154Z","submitted_at":"2026-07-06T03:39:08Z","title":"Enhancing Large Multimodal Models in Key Information Extraction via Scene-Aware Document Synthesis","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-07-11T16:02:00.920066Z"},"links":{"cited_paper":"/paper/2410.05970","citing_paper":"/paper/2607.04636"},"observation_digest":"sha256:91e75dc56b2d477b8a66a779ae8d193d5c333a625696498282314ac58e71f670","observation_id":"f98e27b1-3b68-4450-b015-796f12fdaf20","resolution":{"observed_at":"2026-07-11T16:02:00.920066Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2410.05970/citation-record","integrity":"/paper/2410.05970/integrity","json":"/paper/2410.05970/citation-record.json","paper":"/paper/2410.05970"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2309.08872","last_updated":"2023-11-08T05:09:28Z","snapshot_observed_at":"2026-07-06T16:19:17.917066Z","submitted_at":"2023-09-16T04:29:05Z","title":"PDFTriage: Question Answering over Long, Structured Documents","version":2},"cited_work":{"arxiv_id":"2309.08872","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2309.08872","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Pdftriage: Question answering over long, structured documents","venue":null,"work_id":"7af47451-b7b8-4824-a744-3f87356193c1","year":2023},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"cited_paper":"/paper/2309.08872","citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:4f0c35b161aa249944835f0c4a49b0cfe6ed862a5fd078ea6266c1efc103a81d","observation_id":"59ec3b1c-bf5c-4af9-b826-5472323df84d","resolution":{"observed_at":"2026-05-23T19:43:23.789541Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Prem Jacob, Beatriz Lucia Salvador Bizotto, and Mithi- leysh Sathiyanarayanan","venue":null,"work_id":"acecae93-4dba-4dca-9d57-02db5283de65","year":2024},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:df00371e85b7ddfefc7a36f6e6ee1e466af9c538f0a46793c32caad89396e427","observation_id":"51befefd-d04f-48bf-8eb2-5208b0d833d9","resolution":{"observed_at":"2026-05-23T19:45:48.128531Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"YaRN: Efficient context window extension of large language models","venue":null,"work_id":"bc5303ae-565a-49c4-864c-f6a20179a6b3","year":2024},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:294a60f631888b9dadb8c359e258e203c7acbe4cfff7281a0cbfa488713f74ef","observation_id":"c04e0b88-dd43-4063-9a38-fe2543e1dca6","resolution":{"observed_at":"2026-05-23T19:45:48.124847Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"LongloRA: Efficient fine-tuning of long-context large language models","venue":null,"work_id":"b28b0841-8509-4ee8-a340-d38614cedefc","year":2024},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:26dfa948e4dfc6c1fc9aae3f2045563dd787a797209d9a2cbb2b911e378ad097","observation_id":"62073c19-7e96-4787-84d1-0c74ef39b940","resolution":{"observed_at":"2026-05-23T19:45:48.168960Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Fo- cused transformer: Contrastive training for context scaling","venue":null,"work_id":"eea0c40c-10ef-4a51-ada5-7c000b9193d8","year":2023},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:d462a058fd40a3ac3ef4ef2120b8e9b50f6252893a6ddc1bba2587ec5cb7356c","observation_id":"9396d137-9a68-4429-a27a-cc64ca13ab7c","resolution":{"observed_at":"2026-05-23T19:45:48.149159Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.11325","last_updated":"2023-09-23T18:36:21Z","snapshot_observed_at":"2026-07-06T16:21:17.180009Z","submitted_at":"2023-09-20T13:50:26Z","title":"DISC-LawLLM: Fine-tuning Large Language Models for Intelligent Legal Services","version":2},"cited_work":{"arxiv_id":"2309.11325","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2309.11325","snapshot_observed_at":"2026-07-04T00:59:19.868722Z","title":"Disc-lawllm: Fine-tuning large language models for intelligent legal services","venue":null,"work_id":"065492b2-4be3-4efc-8ef7-592f462732f1","year":2023},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"cited_paper":"/paper/2309.11325","citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:6fb60454c4fe786e2c5fbe4d639759081d4217a45b2e9e96085f47103d666319","observation_id":"93cd4be0-0cf7-438e-b8c0-6dc22e1c589f","resolution":{"observed_at":"2026-05-23T19:43:23.690395Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.15205","last_updated":"2023-10-25T05:56:13Z","snapshot_observed_at":"2026-07-06T16:37:23.707569Z","submitted_at":"2023-10-23T11:33:41Z","title":"DISC-FinLLM: A Chinese Financial Large Language Model based on Multiple Experts Fine-tuning","version":2},"cited_work":{"arxiv_id":"2310.15205","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.15205","snapshot_observed_at":"2026-07-01T10:25:41.504539Z","title":"Disc-finllm: A chinese financial large language model based on multiple experts fine-tuning","venue":null,"work_id":"a793153a-85ce-418e-b576-8e9a88ed22a7","year":2023},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"cited_paper":"/paper/2310.15205","citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:11ab306093524636a6b39e8b47c97213a400a520e763669856de69b381d45822","observation_id":"cfcaecbc-6ccb-485d-80cb-492b650e2912","resolution":{"observed_at":"2026-05-23T19:43:23.672488Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.15391","last_updated":"2024-01-27T11:41:48Z","snapshot_observed_at":"2026-08-02T18:59:01.488566Z","submitted_at":"2024-01-27T11:41:48Z","title":"MultiHop-RAG: Benchmarking Retrieval-Augmented Generation for Multi-Hop Queries","version":1},"cited_work":{"arxiv_id":"2401.15391","doi":"10.48550/arxiv.2401.15391","metadata_source":"pith","pith_arxiv_id":"2401.15391","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MultiHop-RAG: Benchmarking Retrieval-Augmented Generation for Multi-Hop Queries","venue":"cs.CL","work_id":"f7e94a4e-a9ec-4556-8cb5-0f3d0d2d01d4","year":2024},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"cited_paper":"/paper/2401.15391","citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:fa89760e78df15add44b4b534bc64cbc9dfdb6e02c1cd5770c6e3ea87e89af77","observation_id":"44f5dced-529e-4bf6-8c7c-58543959a0ed","resolution":{"observed_at":"2026-05-23T19:43:23.850554Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-13T22:51:19.84429+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-13T22:51:19.84429+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16130","last_updated":"2025-02-19T10:49:41Z","snapshot_observed_at":"2026-07-06T18:05:11.700127Z","submitted_at":"2024-04-24T18:38:11Z","title":"From Local to Global: A Graph RAG Approach to Query-Focused Summarization","version":2},"cited_work":{"arxiv_id":"2404.16130","doi":"10.48550/arxiv.2404.16130","metadata_source":"pith","pith_arxiv_id":"2404.16130","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"From Local to Global: A Graph RAG Approach to Query-Focused Summarization","venue":"cs.CL","work_id":"588618d7-fd41-4053-b34d-a981f8793039","year":2024},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"cited_paper":"/paper/2404.16130","citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:16cb3ac89b55c1870d510eb299daf9d6826de6e3820fc9b1e72f569fe32981fd","observation_id":"7a9822f6-5e06-4260-80e4-9860f2969627","resolution":{"observed_at":"2026-05-23T19:43:23.858023Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T05:49:57.694079+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T05:49:57.694079+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.04473","last_updated":"2024-03-15T06:51:30Z","snapshot_observed_at":"2026-08-04T10:28:00.554323Z","submitted_at":"2024-03-07T13:16:24Z","title":"TextMonkey: An OCR-Free Large Multimodal Model for Understanding Document","version":2},"cited_work":{"arxiv_id":"2403.04473","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.04473","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Textmonkey: An ocr-free large multimodal model for understanding document","venue":null,"work_id":"612fe156-781e-49f3-bac5-03fcb13d115f","year":2024},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"cited_paper":"/paper/2403.04473","citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:eeea753060fdba80e5c63639c269c75d3f6e67dbfd5e5d4cbd949f2b86d99260","observation_id":"8a11461b-8a19-4f64-b893-9cb44861c518","resolution":{"observed_at":"2026-05-23T19:43:23.776809Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16821","last_updated":"2024-04-29T20:24:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-04-25T17:59:19Z","title":"How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites","version":2},"cited_work":{"arxiv_id":"2404.16821","doi":"10.48550/arxiv.2404.16821","metadata_source":"pith","pith_arxiv_id":"2404.16821","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites","venue":"cs.CV","work_id":"3714835e-c5a6-4d7e-950c-be44670ed9e6","year":2024},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"cited_paper":"/paper/2404.16821","citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:0ec31c033cae3c3da1b4b5a62bba08cd01ac76582625c4e41ec813db5e479258","observation_id":"ea104539-ffd2-4e9a-8df9-1f0db8dcca74","resolution":{"observed_at":"2026-05-23T19:43:23.837243Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Vary: Scaling up the vision vocabulary for large vision-language model","venue":null,"work_id":"26337152-436a-40db-9cf2-c4b8875ba5bf","year":2024},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:0ae0ea62aa5115d7df54f6f4315972123ab8670bf93e128340b339305b4f0206","observation_id":"4bfd8580-e6f3-445c-a550-23a40ebe2a42","resolution":{"observed_at":"2026-05-23T19:45:48.005008Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.14295","last_updated":"2024-05-23T08:15:49Z","snapshot_observed_at":"2026-08-06T05:45:07.767677Z","submitted_at":"2024-05-23T08:15:49Z","title":"Focus Anywhere for Fine-grained Multi-page Document Understanding","version":1},"cited_work":{"arxiv_id":"2405.14295","doi":null,"metadata_source":"pith","pith_arxiv_id":"2405.14295","snapshot_observed_at":"2026-07-08T20:35:34.401926Z","title":"Focus anywhere for fine- grained multi-page document understanding","venue":"cs.CV","work_id":"f9c7794d-c8f0-42dd-b0c1-7bba21f80c07","year":2024},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"cited_paper":"/paper/2405.14295","citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:f2373eaee75e1f6d72dc938ccee3e1da1a3acd9d9c1db46e84e7ecb7d80a1dad","observation_id":"24203791-56ad-4de6-a38b-b4b5b4e634cd","resolution":{"observed_at":"2026-05-23T19:43:23.844054Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Hi- erarchical multimodal transformers for multipage docvqa","venue":null,"work_id":"ccaeecd0-1d89-490d-80fc-8f615e5731bb","year":2023},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:4916ed950f6611e41501aa59d8de46d26a5e6c29297e1228ad9ac9da0082f424","observation_id":"2787279b-24bf-4388-ba86-2c12f396c7fe","resolution":{"observed_at":"2026-05-23T19:45:48.011488Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Slidevqa: A dataset for document visual question answering on multiple images","venue":null,"work_id":"04fd730e-3b07-49c3-ab8c-1d26015e862f","year":2023},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:3102d182e57c574e4876de39a6f9b3ceeae3d464834dc40c0b59f6f95fca02c7","observation_id":"ec48b73a-84f4-4f5f-8f27-b883ea709152","resolution":{"observed_at":"2026-05-23T19:45:48.152428Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gram: Global reasoning for multi-page vqa","venue":null,"work_id":"42ebd695-05e8-42c2-a727-faa3854c6d4e","year":null},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:1068d12201d66344f29ec09b2fdee626eee86340ddcc8714d37fd997ded01084","observation_id":"4e3ff36b-f80b-416a-ab1e-3fa47497083c","resolution":{"observed_at":"2026-05-23T19:45:48.155764Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Document understanding dataset and evaluation (dude)","venue":null,"work_id":"bae09ef5-9075-4fa0-b25c-bdb8d586ac9c","year":2023},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:ffa0a031aa5f34b2957946203831858cfbf769b329fdb3e23abb94b71f1eb552","observation_id":"1e4a60e8-f72c-49be-a5a3-4d04c3c4507c","resolution":{"observed_at":"2026-05-23T19:45:48.138882Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.07230","last_updated":"2024-10-09T07:46:02Z","snapshot_observed_at":"2026-07-06T18:28:51.180903Z","submitted_at":"2024-06-11T13:09:16Z","title":"Needle In A Multimodal Haystack","version":2},"cited_work":{"arxiv_id":"2406.07230","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.07230","snapshot_observed_at":"2026-06-30T21:05:04.221956Z","title":"Needle in a multimodal haystack","venue":null,"work_id":"2e2bc244-b863-4659-8cb3-46768e090f83","year":2024},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"cited_paper":"/paper/2406.07230","citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:80e23a91a508fa3a0c3ebc364af033fe24aeae471f18436b5aae201bdecc89f4","observation_id":"14cdd7a0-dc02-4776-bf0d-ddf33a65acab","resolution":{"observed_at":"2026-05-23T19:43:23.758786Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"RAPTOR: Re- cursive abstractive processing for tree-organized retrieval","venue":null,"work_id":"853de292-6090-420c-a632-eeee9e148f46","year":2024},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:c68e1541f6c864ae26bba33895d5e1e00c3366f013ef3fe128e9bf15b8a1b897","observation_id":"8a0fa990-2ef3-42cf-b723-f59f72c3c5ce","resolution":{"observed_at":"2026-05-23T19:45:48.142523Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Unidoc: A univer- sal large multimodal model for simultaneous text detection, recognition, spotting and understanding","venue":null,"work_id":"47f337eb-a8f8-465d-86c4-43adebd930e0","year":2023},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:27ad2e941016905e7624338cf5f9ebac04816d7d135353139aa05d5be41e1db0","observation_id":"5c4d44eb-128a-43f5-8845-ecf38bff1e41","resolution":{"observed_at":"2026-05-23T19:45:48.132108Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.02499","last_updated":"2023-07-04T11:28:07Z","snapshot_observed_at":"2026-08-01T15:24:02.956161Z","submitted_at":"2023-07-04T11:28:07Z","title":"mPLUG-DocOwl: Modularized Multimodal Large Language Model for Document Understanding","version":1},"cited_work":{"arxiv_id":"2307.02499","doi":"10.48550/arxiv.2307.02499","metadata_source":"arxiv_reference","pith_arxiv_id":"2307.02499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"mplug-docowl: Modularized multimodal large language model for document understanding","venue":"arXiv (Cornell University)","work_id":"990d95b3-f658-488d-b1a6-169e9e6aa7fb","year":2023},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"cited_paper":"/paper/2307.02499","citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:0f81cb0b8ed0bcecf269c6e00646204d3d195647269ca18fdbbe16c4c208cf8f","observation_id":"cd5b616a-9db4-4433-8e40-f861a0281528","resolution":{"observed_at":"2026-05-23T19:43:23.728217Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.05126","last_updated":"2023-10-08T11:33:09Z","snapshot_observed_at":"2026-07-06T16:29:23.853368Z","submitted_at":"2023-10-08T11:33:09Z","title":"UReader: Universal OCR-free Visually-situated Language Understanding with Multimodal Large Language Model","version":1},"cited_work":{"arxiv_id":"2310.05126","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2310.05126","snapshot_observed_at":"2026-07-03T16:18:37.543028Z","title":"Ureader: Universal ocr-free visually-situated language understanding with multimodal large language model","venue":null,"work_id":"acc4fea5-7f8a-48c1-b1c9-2389be85b798","year":2023},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"cited_paper":"/paper/2310.05126","citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:3f238899f83356c8a6fc2b36a740fa196d6e5f68d5af1fe95880df3a336cfe6c","observation_id":"dc054965-971e-4bc6-872d-ed373c5e9791","resolution":{"observed_at":"2026-05-23T19:43:23.831310Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Llava-next: Im- 9 proved reasoning, ocr, and world knowledge, January 2024","venue":null,"work_id":"3c6c1161-6eda-486c-aa9b-70ac9131461f","year":2024},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:15bf52069142340aa1fea2a7426ac8e4b3c7034378e572f013c6dd549bcaaea6","observation_id":"0a4f7277-ce3d-4289-a2a8-045c1ca16437","resolution":{"observed_at":"2026-05-23T19:45:48.135201Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.06512","last_updated":"2024-04-09T17:59:32Z","snapshot_observed_at":"2026-07-06T17:57:56.632041Z","submitted_at":"2024-04-09T17:59:32Z","title":"InternLM-XComposer2-4KHD: A Pioneering Large Vision-Language Model Handling Resolutions from 336 Pixels to 4K HD","version":1},"cited_work":{"arxiv_id":"2404.06512","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.06512","snapshot_observed_at":"2026-07-04T08:19:44.167431Z","title":"Internlm-xcomposer2-4khd: A pioneer- ing large vision-language model handling resolutions from 336 pixels to 4k hd","venue":null,"work_id":"799b1f12-2edc-4b09-a83b-dbd87e1404e7","year":2024},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"cited_paper":"/paper/2404.06512","citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:138963c39b5d1945d21f4fff5cbf922ca473fb22f09a4945bb93afdb85db16d0","observation_id":"3e634f4d-3aed-4300-ae02-6f829d41dc71","resolution":{"observed_at":"2026-05-23T19:43:23.812998Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"mplug- docowl2: High-resolution compressing for ocr-free multi- page document understanding","venue":null,"work_id":"64503230-9dbf-4b8e-a750-bb74a10e005e","year":2024},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:72ca850f73560249e17de9abba31a2aa93833c585b40ffae256f01d8d7e2f653","observation_id":"0811788f-b25a-4749-996f-216d0da37f59","resolution":{"observed_at":"2026-05-23T19:45:48.145975Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Cream: Coarse-to- fine retrieval and multi-modal efficient tuning for document vqa","venue":null,"work_id":"f508ce3f-4e51-4177-8968-c327b7d4be61","year":2024},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:f3d0f11d7c18da50ce89455b46f44bb4fe19bf528a8724b5d8d17b7132878fb9","observation_id":"36e89ca5-4f32-40b2-854b-82a831c01e0e","resolution":{"observed_at":"2026-05-23T19:45:48.159121Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Efficient attentions for long document summa- rization","venue":null,"work_id":"8962e5b3-da7e-4b98-8153-87c98277abdb","year":2021},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:452c0b3d20416109a749346cbf6931e9e82ee37026263cda202304f8fe1f3448","observation_id":"844dce37-954d-4756-9ac1-5a32abaeb84d","resolution":{"observed_at":"2026-05-23T19:45:48.162272Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"A dataset of information-seeking questions and answers anchored in research papers","venue":null,"work_id":"6c9fc72f-ec9c-43af-b4c3-7ee79d05e518","year":2021},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:63fcf2c6d2e225000ff21dbccd5031cd026c61ad3b12ddbf379b4d9cb326bc24","observation_id":"088c87cb-557d-4c1c-9bde-ecb7754e71ee","resolution":{"observed_at":"2026-05-23T19:45:48.121380Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Pub- laynet: largest dataset ever for document layout analysis","venue":null,"work_id":"02537605-0b9e-42a0-9494-0d514cb959e3","year":2019},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:967f7f189404a6f1a07e751e0bd3e1c0d672d44d01bbfa53fba6a0216083e235","observation_id":"1b7d594a-7d53-41b2-933e-d1442fd41f19","resolution":{"observed_at":"2026-05-23T19:45:48.114634Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.01038","last_updated":"2020-11-11T05:08:05Z","snapshot_observed_at":"2026-08-05T18:37:17.572177Z","submitted_at":"2020-06-01T16:04:30Z","title":"DocBank: A Benchmark Dataset for Document Layout Analysis","version":3},"cited_work":{"arxiv_id":"2006.01038","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2006.01038","snapshot_observed_at":"2026-06-30T06:54:20.455057Z","title":"Docbank: A bench- mark dataset for document layout analysis","venue":null,"work_id":"61a460a7-76bb-4f4d-9930-64d749fa7721","year":2006},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"cited_paper":"/paper/2006.01038","citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:4ff96c9cc16a3aef71b6453de002a7a34315b57e4226f408f044ed073cbd3bbe","observation_id":"cbe6757a-b898-4a41-8f09-6c9fe10f03f3","resolution":{"observed_at":"2026-05-23T19:43:23.753536Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Doclaynet: a large human-annotated dataset for document-layout segmentation","venue":null,"work_id":"850c0c7d-e145-438c-b197-2892a4a31148","year":2022},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:76e53605a369641cf1f2d8c42060ceec88caffd2c8bc6b2b10234254bb8f0bd6","observation_id":"fcaa281b-b571-42b3-8630-eefd8e131e73","resolution":{"observed_at":"2026-05-23T19:45:48.106182Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Docile benchmark for document information localization and extraction","venue":null,"work_id":"842de72a-515f-4686-843f-ab4b5650127f","year":2023},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:e6e8afcbc9e4dadffada07ff002e64bc050ad1e640bc099cb042e9782716f32a","observation_id":"702c1049-5825-4624-a60c-fcc7d759a399","resolution":{"observed_at":"2026-05-23T19:45:48.187571Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Cord: A con- solidated receipt dataset for post-ocr parsing","venue":null,"work_id":"4cf05b03-0936-471a-9206-a583b5280c03","year":2019},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:815a1e702f392e9a5c1d24dacc4f618bce0ab35dad9c8d941ce97f19e675415e","observation_id":"e152d2b3-749e-4433-ad17-e80cf27fbcc1","resolution":{"observed_at":"2026-05-23T19:45:48.165556Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Icdar2019 com- petition on scanned receipt ocr and information extraction","venue":null,"work_id":"d88d0463-faa3-4735-b84f-86cb44f73f60","year":2019},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:3fba1ee606f1ada70cb699c88e72a822e0ffe247368ca76d7150ba91a1f55396","observation_id":"1bdb4dcc-044e-4e9a-9466-00e5bcfb68fd","resolution":{"observed_at":"2026-05-23T19:45:48.183595Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"a55d790d-3040-4a75-9915-bde56e8d02d9","year":2021},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:6842e525f8635d0e5419de887405cc9d6eb4660b517b7348092a833bfdf8a587","observation_id":"9e40d379-dbc8-4ae5-af87-86a7724d0416","resolution":{"observed_at":"2026-05-23T19:45:48.109798Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Ocr-vqa: Visual question answering by reading text in images","venue":null,"work_id":"d2e2b32d-f2e8-4fe5-9f6f-d871d864cbe7","year":2019},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:e5a03431c8bca01fbead7106e4d2836109eeec0cea362c99e11f09ec5e70dbba","observation_id":"bb13bd1e-9a8b-4239-a274-0d231e470592","resolution":{"observed_at":"2026-05-23T19:45:48.099494Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.10244","last_updated":"2022-03-19T05:00:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-03-19T05:00:30Z","title":"ChartQA: A Benchmark for Question Answering about Charts with Visual and Logical Reasoning","version":1},"cited_work":{"arxiv_id":"2203.10244","doi":null,"metadata_source":"pith","pith_arxiv_id":"2203.10244","snapshot_observed_at":"2026-07-04T16:39:57.602645Z","title":"ChartQA: A Benchmark for Question Answering about Charts with Visual and Logical Reasoning","venue":"cs.CL","work_id":"8b49b78c-7e1d-4f57-af10-7c11cd63ff7c","year":2022},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"cited_paper":"/paper/2203.10244","citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:990f84bfad3f0bfdad2aa24e9468a6d2d4f50bc06050bc5a8e70c5d58d2cf670","observation_id":"a5cc83a2-013a-4aa5-992d-55e3800a39fb","resolution":{"observed_at":"2026-05-23T19:43:23.715241Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.12185","last_updated":"2025-04-27T11:31:23Z","snapshot_observed_at":"2026-07-06T17:32:15.447061Z","submitted_at":"2024-02-19T14:48:23Z","title":"ChartX & ChartVLM: A Versatile Benchmark and Foundation Model for Complicated Chart Reasoning","version":6},"cited_work":{"arxiv_id":"2402.12185","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.12185","snapshot_observed_at":"2026-07-04T16:59:57.579605Z","title":"Chartx & chartvlm: A versatile bench- mark and foundation model for complicated chart reasoning","venue":null,"work_id":"18c69f85-a4de-4a52-920f-b4f5d0dc0508","year":2024},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"cited_paper":"/paper/2402.12185","citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:d8ddbbac0d2bc69ef798cdecf0a6e7b30e20ddc783fa4f490df4af9b0dde877d","observation_id":"b138e72b-54d3-4a0e-bed2-1ff8e8c9b124","resolution":{"observed_at":"2026-05-23T19:43:23.678200Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.00231","last_updated":"2024-06-02T15:47:16Z","snapshot_observed_at":"2026-07-06T17:37:52.514730Z","submitted_at":"2024-03-01T02:21:30Z","title":"Multimodal ArXiv: A Dataset for Improving Scientific Comprehension of Large Vision-Language Models","version":3},"cited_work":{"arxiv_id":"2403.00231","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.00231","snapshot_observed_at":"2026-07-03T20:48:56.198380Z","title":"Multimodal arxiv: A dataset for improving scientific comprehension of large vision-language models","venue":null,"work_id":"4af8a742-b9ce-48eb-99e6-35eb96332561","year":2024},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"cited_paper":"/paper/2403.00231","citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:032797d85e8fe3e43fa3398062dc6338ad66af2cf83410cbf4fe12e4ffc913e8","observation_id":"d7c5483a-670c-4b63-bdef-68f7bf107a24","resolution":{"observed_at":"2026-05-23T19:43:23.783118Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"fb063017-5528-4475-9e9e-7cdf1c708853","year":2022},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:2ec8d2f4827f8a010de0453641d8943d58b292d2a265f58c40d993582b4a1d98","observation_id":"cc4a16dc-a7af-4172-bbd0-fecab8d8d01c","resolution":{"observed_at":"2026-05-23T19:45:48.096550Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12191","last_updated":"2024-10-03T15:54:49Z","snapshot_observed_at":"2026-08-06T05:35:29.109022Z","submitted_at":"2024-09-18T17:59:32Z","title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","version":2},"cited_work":{"arxiv_id":"2409.12191","doi":"10.48550/arxiv.2409.12191","metadata_source":"pith","pith_arxiv_id":"2409.12191","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","venue":"cs.CV","work_id":"8abcfe4f-e0fb-44b7-9123-448fac95f90a","year":2024},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"cited_paper":"/paper/2409.12191","citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:5ce0e0a0dbd4a106b2c807ad7cb63d3b112f411d16f2fba510389775bfcc4e14","observation_id":"9cfde41b-d470-4329-91f8-7a527814720c","resolution":{"observed_at":"2026-05-23T19:43:23.721597Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T02:19:33.884263+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T02:19:33.884263+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2406.11633","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Docgenome: An open large- scale scientific document benchmark for training and test- ing multi-modal large language models","venue":null,"work_id":"a1f0b095-dd56-424c-b917-b40bde568c6d","year":2024},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:a0ccca0a0e9df0b1352e799d58d0ea61399fdbeaa194357b58610f6d5fba56c6","observation_id":"3e3b8f40-4ef9-4520-b393-42e1a00eca51","resolution":{"observed_at":"2026-05-23T19:43:23.740574Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05530","last_updated":"2024-12-16T17:39:39Z","snapshot_observed_at":"2026-07-06T17:41:42.995949Z","submitted_at":"2024-03-08T18:54:20Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","version":5},"cited_work":{"arxiv_id":"2403.05530","doi":"10.48550/arxiv.2403.05530","metadata_source":"pith","pith_arxiv_id":"2403.05530","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","venue":"cs.CL","work_id":"80e3e977-f1bb-4c83-8d0c-1ab0a0c5c3f1","year":2024},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"cited_paper":"/paper/2403.05530","citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:db21d0d957998f49933a1eb8515d44609754beb50a6c28086480dbcf8359fb83","observation_id":"d7c11ba4-317b-46ab-a8ea-093ef6059d9c","resolution":{"observed_at":"2026-05-23T19:43:23.806608Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-02T03:08:14.426583+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-02T03:08:14.426583+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gpt-4v(ision) system card","venue":null,"work_id":"db365ac7-9c75-4de4-a69b-e434718f490e","year":2023},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:9938f7acf35f198a4efe7c7b28b7c0337c70ac4432cdb650ba792a8454d0ee1e","observation_id":"ba688170-4cc9-41be-9210-1982b6be366e","resolution":{"observed_at":"2026-05-23T19:45:48.102491Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03216","last_updated":"2025-12-12T11:26:32Z","snapshot_observed_at":"2026-07-06T17:25:35.493362Z","submitted_at":"2024-02-05T17:26:49Z","title":"M3-Embedding: Multi-Linguality, Multi-Functionality, Multi-Granularity Text Embeddings Through Self-Knowledge Distillation","version":5},"cited_work":{"arxiv_id":"2402.03216","doi":"10.48550/arxiv.2402.03216","metadata_source":"pith","pith_arxiv_id":"2402.03216","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"M3-Embedding: Multi-Linguality, Multi-Functionality, Multi-Granularity Text Embeddings Through Self-Knowledge Distillation","venue":"cs.CL","work_id":"a9435752-4e49-42bd-95b4-0fec975633c8","year":2024},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"cited_paper":"/paper/2402.03216","citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:6f94595c034edf73ee0c8877dedd4f7538c03ade12df1480783edefca9d363cc","observation_id":"af80f1fe-895f-4d73-b21f-44ec17f08942","resolution":{"observed_at":"2026-05-23T19:43:23.770260Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Hi- erarchical multimodal transformers for multipage docvqa","venue":null,"work_id":"b376fb30-19c5-4f51-b221-500448190dc9","year":2023},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:b83db08812330c1de9ef15c9ab2c090e858c893a3257e6b6574387d4981a0f77","observation_id":"d9c5e129-8a67-4f14-86f2-1161c8ecb1e5","resolution":{"observed_at":"2026-05-23T19:45:48.117609Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"https://github.com/kermitt2/grobid, 2008–2024","venue":null,"work_id":"e0677dee-5b18-4afe-ac72-e057a63a6693","year":2008},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:e71b15af994a22fc108593c4b28998f800976ae5dd74156b1f399768601c6d18","observation_id":"0f711f0f-29ac-4e59-8891-f33b65bb980d","resolution":{"observed_at":"2026-05-23T19:45:48.081836Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Mineru: An open-source solution for precise document content extrac- tion","venue":null,"work_id":"7f91a6d7-8214-46d5-a2ef-99c392d9b5c7","year":2024},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:924413f09dc1e094a2cdddcba4ba9139d8af5c938cc411f6fabbbe9b1bff9cdb","observation_id":"0abb7f0e-61cc-4a4b-83f6-8a7587c6d067","resolution":{"observed_at":"2026-05-23T19:45:48.085541Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.16420","last_updated":"2024-01-29T18:59:02Z","snapshot_observed_at":"2026-08-05T03:42:54.599829Z","submitted_at":"2024-01-29T18:59:02Z","title":"InternLM-XComposer2: Mastering Free-form Text-Image Composition and Comprehension in Vision-Language Large Model","version":1},"cited_work":{"arxiv_id":"2401.16420","doi":null,"metadata_source":"pith","pith_arxiv_id":"2401.16420","snapshot_observed_at":"2026-07-04T19:50:11.273770Z","title":"InternLM-XComposer2: Mastering Free-form Text-Image Composition and Comprehension in Vision-Language Large Model","venue":"cs.CV","work_id":"93411487-d575-4b44-9d9e-2374874a66bd","year":2024},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"cited_paper":"/paper/2401.16420","citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:1f8029d22a482bbe216ed7f11667e3b6eeb583f5166d1ee0ba65a558baa58c5f","observation_id":"f0b63aec-932e-42ec-ab88-2ce839bfe3de","resolution":{"observed_at":"2026-05-23T19:43:23.764585Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":"2312.11805","doi":"10.1038/nrn2888","metadata_source":"pith","pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gemini: A Family of Highly Capable Multimodal Models","venue":"cs.CL","work_id":"83f7c85b-3f11-450f-ac0c-64d9745220b2","year":2023},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:53322f2846f9f5a9f73ae4947c27d9f3f491cc395d3d5ff9661dc1d80fc41dce","observation_id":"2361ccb1-027f-4f04-b526-c759b826729e","resolution":{"observed_at":"2026-05-23T19:43:23.824722Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"a8a6bf58-796c-4da4-99c3-e6c385313849","year":null},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:46adf3ae5de4761e250a336eace69a5b611ac0ec0b032b8909b59806c76a3f26","observation_id":"e2029b65-e72e-4344-a28f-76d5685b4cf4","resolution":{"observed_at":"2026-05-23T19:45:48.092881Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12793","last_updated":"2024-07-30T03:58:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-18T16:58:21Z","title":"ChatGLM: A Family of Large Language Models from GLM-130B to GLM-4 All Tools","version":2},"cited_work":{"arxiv_id":"2406.12793","doi":"10.48550/arxiv.2406.12793","metadata_source":"pith","pith_arxiv_id":"2406.12793","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"ChatGLM: A Family of Large Language Models from GLM-130B to GLM-4 All Tools","venue":"cs.CL","work_id":"de9ce5af-0d8d-4b94-9793-64968d9bc06d","year":2024},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"cited_paper":"/paper/2406.12793","citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:5b49429e614d283518357bdf19628f6352425ee4db4c039f404f891f4cd85859","observation_id":"b4e3fcab-1bab-4762-942f-5647ee71f516","resolution":{"observed_at":"2026-05-23T19:43:23.795542Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.10671","last_updated":"2024-09-10T13:25:53Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-07-15T12:35:42Z","title":"Qwen2 Technical Report","version":4},"cited_work":{"arxiv_id":"2407.10671","doi":"10.18653/v1/2024.naacl-long.246","metadata_source":"pith","pith_arxiv_id":"2407.10671","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen2 Technical Report","venue":"cs.CL","work_id":"a1857881-ab9b-4b80-9b5f-9ae4b5c2566d","year":2024},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"cited_paper":"/paper/2407.10671","citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:36490bc754961c55efd286bc45c5e4864237863312422ba6fe4a522ab95ee2f4","observation_id":"22bffec0-53ee-4851-aa26-0bffa56f0da8","resolution":{"observed_at":"2026-05-23T19:43:23.800975Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Layoutlmv3: Pre-training for document ai with unified text and image masking","venue":null,"work_id":"1722d916-1b3c-4b41-9bfe-3b1861e8f6f7","year":null},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:e8ee0290ea0e69ac7f6fb0fead9e6387bf769c8fa7425ad7e6b048c2c4c978a1","observation_id":"83bfedeb-186b-410e-9eb8-7d66e48f2710","resolution":{"observed_at":"2026-05-23T19:45:48.078658Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2004.05150","last_updated":"2020-12-02T17:52:35Z","snapshot_observed_at":"2026-07-31T17:17:17.205582Z","submitted_at":"2020-04-10T17:54:09Z","title":"Longformer: The Long-Document Transformer","version":2},"cited_work":{"arxiv_id":"2004.05150","doi":"10.48550/arxiv.2004.05150","metadata_source":"pith","pith_arxiv_id":"2004.05150","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Longformer: The Long-Document Transformer","venue":"cs.CL","work_id":"abea7a44-6668-4de7-aab6-f53a6e5aa088","year":2020},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"cited_paper":"/paper/2004.05150","citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:246c8017f2f909da87309cc7c456f1d26e63b2664e7b0dc5e098fe0fb19b4c90","observation_id":"c31e0714-37f5-4aed-ab58-a05416c3cda6","resolution":{"observed_at":"2026-05-23T19:43:23.818907Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-12T21:49:59.161233+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T21:49:59.161233+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Big bird: Transformers for longer sequences","venue":null,"work_id":"32e62ea6-b5aa-49e0-8fcd-22e28889d2af","year":2020},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:7bc265aa847ca87f6cda21fd9daebd561f79af67e8104d6d733268287bbf6fc2","observation_id":"3d381cd5-5034-4818-bef2-63117802afe3","resolution":{"observed_at":"2026-05-23T19:45:48.072366Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.12966","last_updated":"2023-10-13T02:41:28Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-24T17:59:17Z","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","version":3},"cited_work":{"arxiv_id":"2308.12966","doi":"10.48550/arxiv.2308.12966","metadata_source":"pith","pith_arxiv_id":"2308.12966","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","venue":"cs.CV","work_id":"cbc2bb21-b6bb-46c0-80bf-107e195ffe10","year":2023},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"cited_paper":"/paper/2308.12966","citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:bfb6d43a52a6d1d46ff0af1cd57e9aac778796bc3bf2aff75359de4eb2f2d053","observation_id":"f5878661-be7c-44c6-b3da-a0f878c8a855","resolution":{"observed_at":"2026-05-23T19:43:23.709537Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.06607","last_updated":"2024-08-26T06:57:51Z","snapshot_observed_at":"2026-08-05T16:46:37.436149Z","submitted_at":"2023-11-11T16:37:41Z","title":"Monkey: Image Resolution and Text Label Are Important Things for Large Multi-modal Models","version":4},"cited_work":{"arxiv_id":"2311.06607","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.06607","snapshot_observed_at":"2026-07-04T06:39:37.479106Z","title":"Mon- key: Image resolution and text label are important things for large multi-modal models","venue":null,"work_id":"1b51b65b-5659-4d2a-b5b3-0a8ac7f88ed5","year":2024},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"cited_paper":"/paper/2311.06607","citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:25f194f535a8be5c35d5b50b2414c44a0b2439442b13228b9b92fb33a9f8ce21","observation_id":"9d3a7fc1-83a7-4d6c-9880-70a9f42332fa","resolution":{"observed_at":"2026-05-23T19:43:23.703956Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Generative multimodal mod- els are in-context learners","venue":null,"work_id":"aae28441-c57f-4ea4-a28d-d97a47ed6929","year":2024},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:12d69885959bfe21b016c9a8007293694c2a017b98d2cdfbc2e056a6921631ef","observation_id":"5471ee76-3e2b-4c07-9a6d-32c2bc3c6d0a","resolution":{"observed_at":"2026-05-23T19:45:48.075483Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.06395","last_updated":"2024-06-03T08:54:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-04-09T15:36:50Z","title":"MiniCPM: Unveiling the Potential of Small Language Models with Scalable Training Strategies","version":3},"cited_work":{"arxiv_id":"2404.06395","doi":"10.48550/arxiv.2404.06395","metadata_source":"pith","pith_arxiv_id":"2404.06395","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MiniCPM: Unveiling the Potential of Small Language Models with Scalable Training Strategies","venue":"cs.CL","work_id":"f20a4304-bd39-414a-923b-d18322e29258","year":2024},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"cited_paper":"/paper/2404.06395","citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:5bdc20792a2b4cb2acd21da6c47a985ec7b01f952e797841df17d6a7c5f9e751","observation_id":"955db559-e0d1-44d1-8684-22cd722d8b81","resolution":{"observed_at":"2026-05-23T19:43:23.746793Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.16500","last_updated":"2024-08-29T12:59:12Z","snapshot_observed_at":"2026-08-05T11:54:14.447608Z","submitted_at":"2024-08-29T12:59:12Z","title":"CogVLM2: Visual Language Models for Image and Video Understanding","version":1},"cited_work":{"arxiv_id":"2408.16500","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.16500","snapshot_observed_at":"2026-07-08T05:34:32.402425Z","title":"CogVLM2: Visual Language Models for Image and Video Understanding","venue":"cs.CV","work_id":"4cd1db02-57ee-4017-9bf2-b9df24e0f9a9","year":2024},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"cited_paper":"/paper/2408.16500","citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:57db6e2b55becf813913b76f177d57bcec120c968fa232c216898aae2f96522a","observation_id":"eb35f914-a198-4dbf-9193-7bd2b7f97aef","resolution":{"observed_at":"2026-05-23T19:43:23.734159Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09204","last_updated":"2024-04-14T09:48:37Z","snapshot_observed_at":"2026-08-01T15:08:25.999360Z","submitted_at":"2024-04-14T09:48:37Z","title":"TextHawk: Exploring Efficient Fine-Grained Perception of Multimodal Large Language Models","version":1},"cited_work":{"arxiv_id":"2404.09204","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.09204","snapshot_observed_at":"2026-06-29T08:13:15.344020Z","title":"Texthawk: Exploring efficient fine- grained perception of multimodal large language models","venue":null,"work_id":"069d74fb-98e3-4699-9ea8-65eb2bafce0e","year":2024},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"cited_paper":"/paper/2404.09204","citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:733a131db3c35d1ffba5ab20d5dd50d84e76faeda5d0435ec6ab1cfae8ed6c11","observation_id":"03c3bda6-d15d-4433-b521-9eefbf563a73","resolution":{"observed_at":"2026-05-23T19:43:23.684066Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Docformerv2: Local features for document understanding","venue":null,"work_id":"36e3e63d-308d-4702-9e50-f5b1c0dac7fa","year":2024},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:71be8bf58df64e254b679eff495f9194f2bfbe662df9cef50b9227a5bb655347","observation_id":"34c9f4d1-ecf7-46ff-838a-a7714a670e9a","resolution":{"observed_at":"2026-05-23T19:45:48.088832Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Obelics: An open web-scale filtered dataset of interleaved image-text documents","venue":null,"work_id":"3e853969-9bfc-49b9-a4f6-b92695907010","year":2024},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:26e231a253ec7373b214c1d90f577a3f89063114e814db999514bf76dded6152","observation_id":"a1f363ee-fd9c-4f4b-9f63-943b70c3ffba","resolution":{"observed_at":"2026-05-23T19:45:48.175909Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Vila: On pre-training for vi- sual language models","venue":null,"work_id":"04c1b113-1d36-4792-b7bf-dbe4587bc728","year":2024},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:4498337288073d65146cbef4ba1a172fd7725a10688dfd6fa26bab2e6ac24953","observation_id":"42287f0c-861a-4685-9553-82626b020e64","resolution":{"observed_at":"2026-05-23T19:45:48.179100Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"https://tongyi.aliyun.com/qianwen/","venue":null,"work_id":"d26cd983-863d-49e7-9c7a-061318eb89ec","year":null},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:5ca481bf56e85233f5c7b3ae98e54b8de00c92e04c53c23a8dee0ec00afc11c5","observation_id":"b9ae670d-aa01-42ae-9bc7-828c9f64695f","resolution":{"observed_at":"2026-05-23T19:45:48.069034Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"https://chatglm.cn/","venue":null,"work_id":"3e967d57-6e29-438e-8c37-4f87e34087e8","year":null},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:270009bb2dfd90194d29585c8f6fb26f3433bb3945cc098fda7c7052f147f62a","observation_id":"2531f673-1e30-463a-94cd-f3a3a7929421","resolution":{"observed_at":"2026-05-23T19:45:48.045505Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"https://kimi.moonshot.cn/","venue":null,"work_id":"7e105d4a-3429-4627-ae7d-215fcf02affb","year":null},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:48b171818d7fe344ac46bbabb92802cbb5f1b9a07cbe92cabbfc1d0fbb0aa81f","observation_id":"73ab3032-ac13-4cd4-b822-26e03913d7db","resolution":{"observed_at":"2026-05-23T19:45:48.062593Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"https://gemini.google.com/","venue":null,"work_id":"e0e7380f-8123-4647-8358-9e78fc4937ce","year":2026},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:42afe07be98f687b8498412f42695def1a0ce63eda32a9fdefd3d8952ba1e41e","observation_id":"a7cca632-fd65-482c-a8ff-0cbcde187078","resolution":{"observed_at":"2026-05-23T19:45:48.057031Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"用于控制电机，实现循迹与避障。 Evidence 1.底层运动系统的软件设计如图5所示，控制核 心是STM32单片机…进入程序后…信息采集完成 后进行数据处理，控制电机相应转动… 2.图5","venue":null,"work_id":"dc58f752-4beb-41e0-a77c-366dcbd0da89","year":null},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:bf81bc85936debc6a6926d29bdcc1f4600bf5633e83547ebea6700c2401e5462","observation_id":"ce7cd5e9-0ee2-45a3-a858-405325b6361d","resolution":{"observed_at":"2026-05-23T19:45:48.029488Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"17f98fe9-9bfa-4b68-abdc-2690db1c315c","year":null},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:b044a78bb1e216e74a83f86c843e7e560bb0c3d8454c00eeab3977d61c7a26ff","observation_id":"6ebc067a-4801-4ac7-abbc-d74500a3668c","resolution":{"observed_at":"2026-05-23T19:45:48.033155Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"98c5a3bb-45a1-4b4f-a3cf-2c46aa264287","year":null},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:9a9dab0b81e4a703d8707645ab39f90f840df6daf1e9711a8c34d5859a805bfc","observation_id":"3e17deaf-cca7-4c39-b34c-7af6ee997b8f","resolution":{"observed_at":"2026-05-23T19:45:48.036969Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"2.图1显示了W-Cu二元相图。在1084°C时，相区标记为“W+Cu”。这意味着在这个 温度下，钨和铜是以各自的固相形式存在的。这一点可以通过浏览图中1084°C线 下的相区标记确认。 Evidence","venue":null,"work_id":"8eb6a7d9-f528-438f-bc50-511e2fd0e864","year":null},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:b0a27e36d0d157338e7cf9b045e1db6e876ed00ae3e5ee496c127c422cb2481b","observation_id":"c518d369-3b1a-44ff-8a42-f9dc42854ce1","resolution":{"observed_at":"2026-05-23T19:45:48.021696Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"b004d8a4-2063-4230-8ae5-f5f76a7dcd10","year":null},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:a74a37cb3dbdf0f0d0868b0bf1a007dc67c6674018b8515c82074df19f2f4d60","observation_id":"a84c9a6e-bc19-4f07-a9a5-adbd8812c2ff","resolution":{"observed_at":"2026-05-23T19:45:48.014653Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"30ef4911-64b6-4e3a-b34f-28894bfe8027","year":null},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:d52474a45cc493c41a114cc576f24863fd5f6d11e8c3179be90bf430eb9e0b58","observation_id":"6803476d-7402-4c38-a96e-f04dbec34176","resolution":{"observed_at":"2026-05-23T19:45:48.018246Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"6adaaf33-a00b-475d-a576-b5c25b997d67","year":null},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:75e5ea6436cf75875ec7b45bf0e053f5735eb8fed24fca7d5e6520d79235ec86","observation_id":"707bda7d-c98f-4915-b4ce-1381ddd174e1","resolution":{"observed_at":"2026-05-23T19:45:48.025433Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"c04ded5b-ad45-4653-8aa7-5bf67a1f0c86","year":null},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:c1da8893fbf8e22ed938020be29249aef0b4f54fe595dd396138f5977f51fef6","observation_id":"8426d14b-11f2-44a3-b39d-d4fff2a42f3c","resolution":{"observed_at":"2026-05-23T19:45:48.040226Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"f2b813f8-7899-4357-8000-503188bdf6ee","year":null},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:1a9e50b8afd382578d7e5a3d7e805a469e9512075c64af4e35d49d25e8c2655b","observation_id":"2586812c-8d1c-460b-92f0-28fb5b57bb7a","resolution":{"observed_at":"2026-05-23T19:45:48.065892Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"51fc56a1-e3cc-4f36-91f1-e3f9c4d7e336","year":null},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:c82e9682607b02ef73f49d8014113d3303cf813fbb101d4d5a7bd213c0a2f7fa","observation_id":"52878216-8501-4bef-a831-087ccb9a52d7","resolution":{"observed_at":"2026-05-23T19:45:48.172434Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"thought chain","venue":null,"work_id":"c779b8b7-74d5-4363-b417-a14c38191931","year":null},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:bad69c0331b239ccffd0f93fa38fa088dd2bd5b8d65c85380b5e4545103781fa","observation_id":"61bfed9c-e1c8-4e4f-a8c6-9cc0d91bf538","resolution":{"observed_at":"2026-05-23T19:45:48.008226Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","latest_version":3,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling"},"reference_resolution":{"displayed":80,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":11,"verified_exact":30,"verified_fuzzy":39},"total_outbound_references":80},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 80 of 80 outbound references and 5 inbound Pith citation observations for arXiv:2410.05970."}