{"as_of":"2026-08-06T16:15:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:e13a4beaa66fed3ae603c927857e127857f25a8554692987b3f68b2b7df09a34","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":21,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":21,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":21,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":21,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T15:57:02.134795Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T06:39:37.479106Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2311.06607","last_updated":"2024-08-26T06:57:51Z","snapshot_observed_at":"2026-08-05T16:46:37.436149Z","submitted_at":"2023-11-11T16:37:41Z","title":"Monkey: Image Resolution and Text Label Are Important Things for Large Multi-modal Models","version":4},"cited_work":{"arxiv_id":"2311.06607","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.06607","snapshot_observed_at":"2026-07-04T06:39:37.479106Z","title":"Mon- key: Image resolution and text label are important things for large multi-modal models","venue":null,"work_id":"1b51b65b-5659-4d2a-b5b3-0a8ac7f88ed5","year":2024},"citing_paper":{"arxiv_id":"2306.13549","last_updated":"2024-11-29T15:51:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-23T15:21:52Z","title":"A Survey on Multimodal Large Language Models","version":4},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-16T02:56:41.658658Z"},"links":{"cited_paper":"/paper/2311.06607","citing_paper":"/paper/2306.13549"},"observation_digest":"sha256:610285715d9287c75df434c47289686e6706cddf68c9c2c5bdc6a794acd53f12","observation_id":"0cb96e4f-8e4b-4928-83dd-191236d49a52","resolution":{"observed_at":"2026-05-16T02:56:42.471878Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.06607","last_updated":"2024-08-26T06:57:51Z","snapshot_observed_at":"2026-08-05T16:46:37.436149Z","submitted_at":"2023-11-11T16:37:41Z","title":"Monkey: Image Resolution and Text Label Are Important Things for Large Multi-modal Models","version":4},"cited_work":{"arxiv_id":"2311.06607","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.06607","snapshot_observed_at":"2026-07-04T06:39:37.479106Z","title":"Mon- key: Image resolution and text label are important things for large multi-modal models","venue":null,"work_id":"1b51b65b-5659-4d2a-b5b3-0a8ac7f88ed5","year":2024},"citing_paper":{"arxiv_id":"2307.06281","last_updated":"2024-08-20T03:56:03Z","snapshot_observed_at":"2026-07-06T15:53:19.485466Z","submitted_at":"2023-07-12T16:23:09Z","title":"MMBench: Is Your Multi-modal Model an All-around Player?","version":5},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-12T17:20:53.687692Z"},"links":{"cited_paper":"/paper/2311.06607","citing_paper":"/paper/2307.06281"},"observation_digest":"sha256:b618e77d0dd6cf3db4a87f5c00172d434397a3231068279379442b309bd96eb9","observation_id":"77798ffe-5e13-4a9e-9b18-b33c9ab5be2a","resolution":{"observed_at":"2026-05-12T17:20:53.896871Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.06607","last_updated":"2024-08-26T06:57:51Z","snapshot_observed_at":"2026-08-05T16:46:37.436149Z","submitted_at":"2023-11-11T16:37:41Z","title":"Monkey: Image Resolution and Text Label Are Important Things for Large Multi-modal Models","version":4},"cited_work":{"arxiv_id":"2311.06607","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.06607","snapshot_observed_at":"2026-07-04T06:39:37.479106Z","title":"Mon- key: Image resolution and text label are important things for large multi-modal models","venue":null,"work_id":"1b51b65b-5659-4d2a-b5b3-0a8ac7f88ed5","year":2024},"citing_paper":{"arxiv_id":"2312.14238","last_updated":"2024-01-15T15:23:55Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-21T18:59:31Z","title":"InternVL: Scaling up Vision Foundation Models and Aligning for Generic Visual-Linguistic Tasks","version":3},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-05-13T22:46:09.693156Z"},"links":{"cited_paper":"/paper/2311.06607","citing_paper":"/paper/2312.14238"},"observation_digest":"sha256:c7c0754757468a555e66db27a58cd5725b9cf55c73e84783c8826c2e9f0c7686","observation_id":"2dc1f223-840d-47b3-83b6-30d50bd8bf64","resolution":{"observed_at":"2026-05-13T22:46:10.003085Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.06607","last_updated":"2024-08-26T06:57:51Z","snapshot_observed_at":"2026-08-05T16:46:37.436149Z","submitted_at":"2023-11-11T16:37:41Z","title":"Monkey: Image Resolution and Text Label Are Important Things for Large Multi-modal Models","version":4},"cited_work":{"arxiv_id":"2311.06607","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.06607","snapshot_observed_at":"2026-07-04T06:39:37.479106Z","title":"Mon- key: Image resolution and text label are important things for large multi-modal models","venue":null,"work_id":"1b51b65b-5659-4d2a-b5b3-0a8ac7f88ed5","year":2024},"citing_paper":{"arxiv_id":"2401.16420","last_updated":"2024-01-29T18:59:02Z","snapshot_observed_at":"2026-08-05T03:42:54.599829Z","submitted_at":"2024-01-29T18:59:02Z","title":"InternLM-XComposer2: Mastering Free-form Text-Image Composition and Comprehension in Vision-Language Large Model","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-17T05:30:27.667126Z"},"links":{"cited_paper":"/paper/2311.06607","citing_paper":"/paper/2401.16420"},"observation_digest":"sha256:4d118abde02adf8403a74e7aafb3ec0f44baea0696a7f8068b1121cd753efb7a","observation_id":"7194c1c2-9fc8-4a1f-91ef-9ce83b2e3d93","resolution":{"observed_at":"2026-05-17T05:30:27.732796Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.06607","last_updated":"2024-08-26T06:57:51Z","snapshot_observed_at":"2026-08-05T16:46:37.436149Z","submitted_at":"2023-11-11T16:37:41Z","title":"Monkey: Image Resolution and Text Label Are Important Things for Large Multi-modal Models","version":4},"cited_work":{"arxiv_id":"2311.06607","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.06607","snapshot_observed_at":"2026-07-04T06:39:37.479106Z","title":"Mon- key: Image resolution and text label are important things for large multi-modal models","venue":null,"work_id":"1b51b65b-5659-4d2a-b5b3-0a8ac7f88ed5","year":2024},"citing_paper":{"arxiv_id":"2402.00253","last_updated":"2024-05-06T01:10:01Z","snapshot_observed_at":"2026-07-06T17:23:26.913214Z","submitted_at":"2024-02-01T00:33:21Z","title":"A Survey on Hallucination in Large Vision-Language Models","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-13T22:10:10.186950Z"},"links":{"cited_paper":"/paper/2311.06607","citing_paper":"/paper/2402.00253"},"observation_digest":"sha256:0cca9181be15efc94fc29cf3fab1fdeafb4ed0aa2f22e3a6ecfca62815732398","observation_id":"d49e1eb5-1728-405e-ba7e-138d7f79f98a","resolution":{"observed_at":"2026-05-13T22:10:10.326163Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.06607","last_updated":"2024-08-26T06:57:51Z","snapshot_observed_at":"2026-08-05T16:46:37.436149Z","submitted_at":"2023-11-11T16:37:41Z","title":"Monkey: Image Resolution and Text Label Are Important Things for Large Multi-modal Models","version":4},"cited_work":{"arxiv_id":"2311.06607","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.06607","snapshot_observed_at":"2026-07-04T06:39:37.479106Z","title":"Mon- key: Image resolution and text label are important things for large multi-modal models","venue":null,"work_id":"1b51b65b-5659-4d2a-b5b3-0a8ac7f88ed5","year":2024},"citing_paper":{"arxiv_id":"2403.09611","last_updated":"2024-04-18T18:51:04Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-03-14T17:51:32Z","title":"MM1: Methods, Analysis & Insights from Multimodal LLM Pre-training","version":4},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-05-16T04:09:36.019146Z"},"links":{"cited_paper":"/paper/2311.06607","citing_paper":"/paper/2403.09611"},"observation_digest":"sha256:0209b577757592d94639caa070ed558de1efd4d3a0ec278843b90037c75a2c21","observation_id":"6b69ddb0-f51d-49e8-ad2a-daef9811f7b1","resolution":{"observed_at":"2026-05-16T04:09:36.493562Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.06607","last_updated":"2024-08-26T06:57:51Z","snapshot_observed_at":"2026-08-05T16:46:37.436149Z","submitted_at":"2023-11-11T16:37:41Z","title":"Monkey: Image Resolution and Text Label Are Important Things for Large Multi-modal Models","version":4},"cited_work":{"arxiv_id":"2311.06607","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.06607","snapshot_observed_at":"2026-07-04T06:39:37.479106Z","title":"Mon- key: Image resolution and text label are important things for large multi-modal models","venue":null,"work_id":"1b51b65b-5659-4d2a-b5b3-0a8ac7f88ed5","year":2024},"citing_paper":{"arxiv_id":"2403.20330","last_updated":"2024-04-09T15:17:50Z","snapshot_observed_at":"2026-08-02T21:55:15.857183Z","submitted_at":"2024-03-29T17:59:34Z","title":"Are We on the Right Way for Evaluating Large Vision-Language Models?","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-12T19:41:44.263663Z"},"links":{"cited_paper":"/paper/2311.06607","citing_paper":"/paper/2403.20330"},"observation_digest":"sha256:fd404cb84072ad597742b06c8efc9a0f80ffb6e93ea4467de33440aa3228bbee","observation_id":"245f14d7-b4f0-49f3-ad45-72fc13368c17","resolution":{"observed_at":"2026-05-12T19:41:44.535674Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.06607","last_updated":"2024-08-26T06:57:51Z","snapshot_observed_at":"2026-08-05T16:46:37.436149Z","submitted_at":"2023-11-11T16:37:41Z","title":"Monkey: Image Resolution and Text Label Are Important Things for Large Multi-modal Models","version":4},"cited_work":{"arxiv_id":"2311.06607","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.06607","snapshot_observed_at":"2026-07-04T06:39:37.479106Z","title":"Mon- key: Image resolution and text label are important things for large multi-modal models","venue":null,"work_id":"1b51b65b-5659-4d2a-b5b3-0a8ac7f88ed5","year":2024},"citing_paper":{"arxiv_id":"2404.16821","last_updated":"2024-04-29T20:24:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-04-25T17:59:19Z","title":"How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites","version":2},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-12T20:58:58.849040Z"},"links":{"cited_paper":"/paper/2311.06607","citing_paper":"/paper/2404.16821"},"observation_digest":"sha256:f43264ff11f6475b06a3c91f5b1009e742cef770f785428b129cf163ef3bc7e3","observation_id":"1ed97519-ff46-4555-8a73-4419df871bfe","resolution":{"observed_at":"2026-05-12T20:58:59.053339Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.06607","last_updated":"2024-08-26T06:57:51Z","snapshot_observed_at":"2026-08-05T16:46:37.436149Z","submitted_at":"2023-11-11T16:37:41Z","title":"Monkey: Image Resolution and Text Label Are Important Things for Large Multi-modal Models","version":4},"cited_work":{"arxiv_id":"2311.06607","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.06607","snapshot_observed_at":"2026-07-04T06:39:37.479106Z","title":"Mon- key: Image resolution and text label are important things for large multi-modal models","venue":null,"work_id":"1b51b65b-5659-4d2a-b5b3-0a8ac7f88ed5","year":2024},"citing_paper":{"arxiv_id":"2407.03320","last_updated":"2024-07-03T17:59:21Z","snapshot_observed_at":"2026-08-04T22:09:42.241578Z","submitted_at":"2024-07-03T17:59:21Z","title":"InternLM-XComposer-2.5: A Versatile Large Vision Language Model Supporting Long-Contextual Input and Output","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-05-17T10:46:28.447347Z"},"links":{"cited_paper":"/paper/2311.06607","citing_paper":"/paper/2407.03320"},"observation_digest":"sha256:4d1a01f7198574c9a674b70ee4ef6326c06264cc540c513f89ae139abe1cc45d","observation_id":"013f2f03-743b-49cf-bbf8-339868f8f464","resolution":{"observed_at":"2026-05-17T10:46:28.688589Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.06607","last_updated":"2024-08-26T06:57:51Z","snapshot_observed_at":"2026-08-05T16:46:37.436149Z","submitted_at":"2023-11-11T16:37:41Z","title":"Monkey: Image Resolution and Text Label Are Important Things for Large Multi-modal Models","version":4},"cited_work":{"arxiv_id":"2311.06607","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.06607","snapshot_observed_at":"2026-07-04T06:39:37.479106Z","title":"Mon- key: Image resolution and text label are important things for large multi-modal models","venue":null,"work_id":"1b51b65b-5659-4d2a-b5b3-0a8ac7f88ed5","year":2024},"citing_paper":{"arxiv_id":"2408.13257","last_updated":"2025-02-05T08:44:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-23T17:59:51Z","title":"MME-RealWorld: Could Your Multimodal LLM Challenge High-Resolution Real-World Scenarios that are Difficult for Humans?","version":3},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-16T07:59:32.638758Z"},"links":{"cited_paper":"/paper/2311.06607","citing_paper":"/paper/2408.13257"},"observation_digest":"sha256:7ce32e7d749876aa4f6f20d9eb756d046d64b6274e97b8861527d932cfe7eb0a","observation_id":"e2c0c614-df4b-4e47-8fa7-185b0efbc18e","resolution":{"observed_at":"2026-05-16T07:59:32.808098Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.06607","last_updated":"2024-08-26T06:57:51Z","snapshot_observed_at":"2026-08-05T16:46:37.436149Z","submitted_at":"2023-11-11T16:37:41Z","title":"Monkey: Image Resolution and Text Label Are Important Things for Large Multi-modal Models","version":4},"cited_work":{"arxiv_id":"2311.06607","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.06607","snapshot_observed_at":"2026-07-04T06:39:37.479106Z","title":"Mon- key: Image resolution and text label are important things for large multi-modal models","venue":null,"work_id":"1b51b65b-5659-4d2a-b5b3-0a8ac7f88ed5","year":2024},"citing_paper":{"arxiv_id":"2410.05970","last_updated":"2026-04-27T13:16:40Z","snapshot_observed_at":"2026-07-06T19:29:42.190813Z","submitted_at":"2024-10-08T12:17:42Z","title":"PDF-WuKong: A Large Multimodal Model for Efficient Long PDF Reading with End-to-End Sparse Sampling","version":3},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-05-23T19:39:35.147671Z"},"links":{"cited_paper":"/paper/2311.06607","citing_paper":"/paper/2410.05970"},"observation_digest":"sha256:25f194f535a8be5c35d5b50b2414c44a0b2439442b13228b9b92fb33a9f8ce21","observation_id":"9d3a7fc1-83a7-4d6c-9880-70a9f42332fa","resolution":{"observed_at":"2026-05-23T19:43:23.703956Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.06607","last_updated":"2024-08-26T06:57:51Z","snapshot_observed_at":"2026-08-05T16:46:37.436149Z","submitted_at":"2023-11-11T16:37:41Z","title":"Monkey: Image Resolution and Text Label Are Important Things for Large Multi-modal Models","version":4},"cited_work":{"arxiv_id":"2311.06607","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.06607","snapshot_observed_at":"2026-07-04T06:39:37.479106Z","title":"Mon- key: Image resolution and text label are important things for large multi-modal models","venue":null,"work_id":"1b51b65b-5659-4d2a-b5b3-0a8ac7f88ed5","year":2024},"citing_paper":{"arxiv_id":"2412.05271","last_updated":"2025-09-26T12:52:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-06T18:57:08Z","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","version":5},"reference_index":140,"source":"pdf_text","source_observed_at":"2026-05-10T13:23:57.588851Z"},"links":{"cited_paper":"/paper/2311.06607","citing_paper":"/paper/2412.05271"},"observation_digest":"sha256:2e455839cc2134c11f9f4d01bf0275ed78b394de023630c9730a725f1c8ee0dc","observation_id":"27a21b8e-3b2d-4ad6-8b6b-c8b3cfe7171e","resolution":{"observed_at":"2026-05-10T13:23:58.253757Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.06607","last_updated":"2024-08-26T06:57:51Z","snapshot_observed_at":"2026-08-05T16:46:37.436149Z","submitted_at":"2023-11-11T16:37:41Z","title":"Monkey: Image Resolution and Text Label Are Important Things for Large Multi-modal Models","version":4},"cited_work":{"arxiv_id":"2311.06607","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.06607","snapshot_observed_at":"2026-07-04T06:39:37.479106Z","title":"Mon- key: Image resolution and text label are important things for large multi-modal models","venue":null,"work_id":"1b51b65b-5659-4d2a-b5b3-0a8ac7f88ed5","year":2024},"citing_paper":{"arxiv_id":"2504.09925","last_updated":"2026-04-29T06:12:36Z","snapshot_observed_at":"2026-08-02T07:57:37.201421Z","submitted_at":"2025-04-14T06:33:29Z","title":"FLARE: Fully Integration of Vision-Language Representations for Deep Cross-Modal Understanding","version":3},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-22T19:49:00.961388Z"},"links":{"cited_paper":"/paper/2311.06607","citing_paper":"/paper/2504.09925"},"observation_digest":"sha256:09fbd7c24bb42ee0c460f7572b10ef42ee5a8caab8fe9af941dd5c5e7ec7b1db","observation_id":"ea2a5ef6-e5f0-4db7-bc73-a67dbf644f8d","resolution":{"observed_at":"2026-05-22T19:52:01.854770Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.06607","last_updated":"2024-08-26T06:57:51Z","snapshot_observed_at":"2026-08-05T16:46:37.436149Z","submitted_at":"2023-11-11T16:37:41Z","title":"Monkey: Image Resolution and Text Label Are Important Things for Large Multi-modal Models","version":4},"cited_work":{"arxiv_id":"2311.06607","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.06607","snapshot_observed_at":"2026-07-04T06:39:37.479106Z","title":"Mon- key: Image resolution and text label are important things for large multi-modal models","venue":null,"work_id":"1b51b65b-5659-4d2a-b5b3-0a8ac7f88ed5","year":2024},"citing_paper":{"arxiv_id":"2504.10479","last_updated":"2025-04-19T03:47:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-14T17:59:25Z","title":"InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models","version":3},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-05-10T13:41:07.991012Z"},"links":{"cited_paper":"/paper/2311.06607","citing_paper":"/paper/2504.10479"},"observation_digest":"sha256:f03a3f186c2ae5dd63ef287a434bb0ebd12fb503adbcfa0c3a631c62359ed9dc","observation_id":"3984b339-064a-43db-9912-503af55cc6e9","resolution":{"observed_at":"2026-05-10T13:41:08.362333Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.06607","last_updated":"2024-08-26T06:57:51Z","snapshot_observed_at":"2026-08-05T16:46:37.436149Z","submitted_at":"2023-11-11T16:37:41Z","title":"Monkey: Image Resolution and Text Label Are Important Things for Large Multi-modal Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.06607","snapshot_observed_at":"2026-08-06T15:57:02.134795Z","title":"Mon- key: Image resolution and text label are important things for large multi-modal models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.14675","last_updated":"2025-07-19T16:03:34Z","snapshot_observed_at":"2026-08-06T15:47:53.180347Z","submitted_at":"2025-07-19T16:03:34Z","title":"Docopilot: Improving Multimodal Models for Document-Level Understanding","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T15:57:02.134795Z"},"links":{"cited_paper":"/paper/2311.06607","citing_paper":"/paper/2507.14675"},"observation_digest":"sha256:e656318137f974fa02c62ecde03a95d781d655bec28ca24ba302d70c9fdb8ffa","observation_id":"543ae786-de83-4dbb-b149-4ecbc8f6fb43","resolution":{"observed_at":"2026-08-06T15:57:02.134795Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.06607","last_updated":"2024-08-26T06:57:51Z","snapshot_observed_at":"2026-08-05T16:46:37.436149Z","submitted_at":"2023-11-11T16:37:41Z","title":"Monkey: Image Resolution and Text Label Are Important Things for Large Multi-modal Models","version":4},"cited_work":{"arxiv_id":"2311.06607","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.06607","snapshot_observed_at":"2026-07-04T06:39:37.479106Z","title":"Mon- key: Image resolution and text label are important things for large multi-modal models","venue":null,"work_id":"1b51b65b-5659-4d2a-b5b3-0a8ac7f88ed5","year":2024},"citing_paper":{"arxiv_id":"2512.10362","last_updated":"2026-04-27T05:32:18Z","snapshot_observed_at":"2026-07-06T22:38:40.044448Z","submitted_at":"2025-12-11T07:22:54Z","title":"Visual Funnel: Resolving Contextual Blindness in Multimodal Large Language Models","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-16T23:47:08.562575Z"},"links":{"cited_paper":"/paper/2311.06607","citing_paper":"/paper/2512.10362"},"observation_digest":"sha256:891219de9ba69c931beff9e414c82eeb3ebd2cfed1c927a08da2d9b73c81027c","observation_id":"88447597-4568-469d-a634-9e20ed3c5b47","resolution":{"observed_at":"2026-05-16T23:48:41.931321Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.06607","last_updated":"2024-08-26T06:57:51Z","snapshot_observed_at":"2026-08-05T16:46:37.436149Z","submitted_at":"2023-11-11T16:37:41Z","title":"Monkey: Image Resolution and Text Label Are Important Things for Large Multi-modal Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.06607","snapshot_observed_at":"2026-08-02T19:58:19.334924Z","title":"Monkey: Image resolution and text label are important things for large multi-modal models.arXiv preprint arXiv:2311.06607, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2603.00461","last_updated":"2026-06-10T06:57:00Z","snapshot_observed_at":"2026-08-05T02:58:46.928689Z","submitted_at":"2026-02-28T04:42:34Z","title":"ReMoT: Reinforcement Learning with Motion Contrast Triplets","version":3},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-02T19:58:19.334924Z"},"links":{"cited_paper":"/paper/2311.06607","citing_paper":"/paper/2603.00461"},"observation_digest":"sha256:2a4df78160186f59695a773db3d431ec4687888fe48da6dbc3cb353dbaf70c2b","observation_id":"57eecde9-6e1e-4448-9fe3-a94055097536","resolution":{"observed_at":"2026-08-02T19:58:19.334924Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.06607","last_updated":"2024-08-26T06:57:51Z","snapshot_observed_at":"2026-08-05T16:46:37.436149Z","submitted_at":"2023-11-11T16:37:41Z","title":"Monkey: Image Resolution and Text Label Are Important Things for Large Multi-modal Models","version":4},"cited_work":{"arxiv_id":"2311.06607","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.06607","snapshot_observed_at":"2026-07-04T06:39:37.479106Z","title":"Mon- key: Image resolution and text label are important things for large multi-modal models","venue":null,"work_id":"1b51b65b-5659-4d2a-b5b3-0a8ac7f88ed5","year":2024},"citing_paper":{"arxiv_id":"2606.17030","last_updated":"2026-06-17T13:54:57Z","snapshot_observed_at":"2026-08-01T21:48:31.832288Z","submitted_at":"2026-06-15T17:52:31Z","title":"Qwen-RobotWorld Technical Report: Unifying Embodied World Modeling through Language-Conditioned Video Generation","version":3},"reference_index":166,"source":"arxiv_source","source_observed_at":"2026-06-27T04:19:26.332718Z"},"links":{"cited_paper":"/paper/2311.06607","citing_paper":"/paper/2606.17030"},"observation_digest":"sha256:91b818dd011c51810840c7b7ddc9dd19b8952a143e9a0036f10713b6b2dfc62b","observation_id":"e15e1f92-b278-4d04-89e0-b91bb83ed2e0","resolution":{"observed_at":"2026-07-03T17:18:43.922398Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.06607","last_updated":"2024-08-26T06:57:51Z","snapshot_observed_at":"2026-08-05T16:46:37.436149Z","submitted_at":"2023-11-11T16:37:41Z","title":"Monkey: Image Resolution and Text Label Are Important Things for Large Multi-modal Models","version":4},"cited_work":{"arxiv_id":"2311.06607","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.06607","snapshot_observed_at":"2026-07-04T06:39:37.479106Z","title":"Mon- key: Image resolution and text label are important things for large multi-modal models","venue":null,"work_id":"1b51b65b-5659-4d2a-b5b3-0a8ac7f88ed5","year":2024},"citing_paper":{"arxiv_id":"2606.21734","last_updated":"2026-06-19T20:43:49Z","snapshot_observed_at":"2026-08-05T18:05:51.515234Z","submitted_at":"2026-06-19T20:43:49Z","title":"HPP: Hierarchical Programmatic Probing for Long Video Understanding by Decoupling Perception and Reasoning","version":1},"reference_index":279,"source":"arxiv_source","source_observed_at":"2026-06-26T14:19:53.450263Z"},"links":{"cited_paper":"/paper/2311.06607","citing_paper":"/paper/2606.21734"},"observation_digest":"sha256:b8c97564520431d62b19c256a2f0fd76568c5202b44856a6f949bef863175607","observation_id":"0ed651f1-a7c3-44ae-8b8a-c60a789c91ac","resolution":{"observed_at":"2026-07-04T06:39:37.480709Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.06607","last_updated":"2024-08-26T06:57:51Z","snapshot_observed_at":"2026-08-05T16:46:37.436149Z","submitted_at":"2023-11-11T16:37:41Z","title":"Monkey: Image Resolution and Text Label Are Important Things for Large Multi-modal Models","version":4},"cited_work":{"arxiv_id":"2311.06607","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.06607","snapshot_observed_at":"2026-07-04T06:39:37.479106Z","title":"Mon- key: Image resolution and text label are important things for large multi-modal models","venue":null,"work_id":"1b51b65b-5659-4d2a-b5b3-0a8ac7f88ed5","year":2024},"citing_paper":{"arxiv_id":"2607.02089","last_updated":"2026-07-01T14:25:43Z","snapshot_observed_at":"2026-08-02T16:57:02.105465Z","submitted_at":"2026-07-01T14:25:43Z","title":"ESC: Emotional Self-Correction for Reliable Vision-Language Models","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-07-03T21:20:00.041277Z"},"links":{"cited_paper":"/paper/2311.06607","citing_paper":"/paper/2607.02089"},"observation_digest":"sha256:a6b7557acb3fa7db07a5948fad93bcc21c4943dc78a57ea19013ac5220de9702","observation_id":"a6cd8e01-3f80-4ee3-86ce-87e3715168a5","resolution":{"observed_at":"2026-07-03T21:28:58.426378Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.06607","last_updated":"2024-08-26T06:57:51Z","snapshot_observed_at":"2026-08-05T16:46:37.436149Z","submitted_at":"2023-11-11T16:37:41Z","title":"Monkey: Image Resolution and Text Label Are Important Things for Large Multi-modal Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.06607","snapshot_observed_at":"2026-07-14T03:31:19.309532Z","title":"arXiv:2311.06607 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.11738","last_updated":"2026-07-13T16:00:03Z","snapshot_observed_at":"2026-08-06T12:36:13.363835Z","submitted_at":"2026-07-13T16:00:03Z","title":"Qwen-Audio-VAE Technical Report","version":1},"reference_index":179,"source":"arxiv_source","source_observed_at":"2026-07-14T03:31:19.309532Z"},"links":{"cited_paper":"/paper/2311.06607","citing_paper":"/paper/2607.11738"},"observation_digest":"sha256:b334c8a7c92bea43162c65d348897a0a5eb001efe7d0980782bb7cae84c15d1c","observation_id":"f03fc67c-d05c-4124-ba0c-b89f0ee94c29","resolution":{"observed_at":"2026-07-14T03:31:19.309532Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2311.06607/citation-record","integrity":"/paper/2311.06607/integrity","json":"/paper/2311.06607/citation-record.json","paper":"/paper/2311.06607"},"outbound":[],"paper":{"arxiv_id":"2311.06607","last_updated":"2024-08-26T06:57:51Z","latest_version":4,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-05T16:46:37.436149Z","submitted_at":"2023-11-11T16:37:41Z","title":"Monkey: Image Resolution and Text Label Are Important Things for Large Multi-modal Models"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 21 inbound Pith citation observations for arXiv:2311.06607."}