{"as_of":"2026-08-10T03:52:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:9a5d85fa786ef01cae9e8e92a050562f676dbf96fc2a7f173b8c75f1383ff520","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":33,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":33,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":33,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":33,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:45:32.061891Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":7,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2401.06209","last_updated":"2024-04-25T07:12:39Z","snapshot_observed_at":"2026-07-06T17:14:32.455890Z","submitted_at":"2024-01-11T18:58:36Z","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2401.06209","doi":"10.48550/arxiv.2401.06209","metadata_source":"arxiv_reference","pith_arxiv_id":"2401.06209","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Eyes wide shut? exploring the visual shortcomings of multimodal llms","venue":"arXiv (Cornell University)","work_id":"fced738d-327a-4c32-b51c-e6df45b47786","year":2024},"citing_paper":{"arxiv_id":"2403.05525","last_updated":"2024-03-11T16:47:41Z","snapshot_observed_at":"2026-08-05T16:51:32.094151Z","submitted_at":"2024-03-08T18:46:00Z","title":"DeepSeek-VL: Towards Real-World Vision-Language Understanding","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-11T17:58:54.177359Z"},"links":{"cited_paper":"/paper/2401.06209","citing_paper":"/paper/2403.05525"},"observation_digest":"sha256:3cc691140463986c7a2a0d41d6bf940e47ce0c10ee2817f5a10e6fa7a22123c3","observation_id":"acac3d1d-5e78-403e-9081-b9765ca49f05","resolution":{"observed_at":"2026-05-11T17:58:54.709008Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06209","last_updated":"2024-04-25T07:12:39Z","snapshot_observed_at":"2026-07-06T17:14:32.455890Z","submitted_at":"2024-01-11T18:58:36Z","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2401.06209","doi":"10.48550/arxiv.2401.06209","metadata_source":"arxiv_reference","pith_arxiv_id":"2401.06209","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Eyes wide shut? exploring the visual shortcomings of multimodal llms","venue":"arXiv (Cornell University)","work_id":"fced738d-327a-4c32-b51c-e6df45b47786","year":2024},"citing_paper":{"arxiv_id":"2403.09611","last_updated":"2024-04-18T18:51:04Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-03-14T17:51:32Z","title":"MM1: Methods, Analysis & Insights from Multimodal LLM Pre-training","version":4},"reference_index":108,"source":"pdf_text","source_observed_at":"2026-05-16T04:09:36.019146Z"},"links":{"cited_paper":"/paper/2401.06209","citing_paper":"/paper/2403.09611"},"observation_digest":"sha256:34169ed6d3672e3073ab9b13731e4fcdda2b92016215394f1b1a7571a73026b8","observation_id":"939e1045-4e5a-4cc2-8d49-572279dcc5e2","resolution":{"observed_at":"2026-05-16T04:09:36.274795Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06209","last_updated":"2024-04-25T07:12:39Z","snapshot_observed_at":"2026-07-06T17:14:32.455890Z","submitted_at":"2024-01-11T18:58:36Z","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2401.06209","doi":"10.48550/arxiv.2401.06209","metadata_source":"arxiv_reference","pith_arxiv_id":"2401.06209","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Eyes wide shut? exploring the visual shortcomings of multimodal llms","venue":"arXiv (Cornell University)","work_id":"fced738d-327a-4c32-b51c-e6df45b47786","year":2024},"citing_paper":{"arxiv_id":"2404.16821","last_updated":"2024-04-29T20:24:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-04-25T17:59:19Z","title":"How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites","version":2},"reference_index":111,"source":"pdf_text","source_observed_at":"2026-05-12T20:58:58.849040Z"},"links":{"cited_paper":"/paper/2401.06209","citing_paper":"/paper/2404.16821"},"observation_digest":"sha256:cc77126ca5352513063d901bbe3fc253eee6740ac16e467c0f82bad829629d51","observation_id":"62c29b44-7eb0-47e4-91b2-6be587e15c2a","resolution":{"observed_at":"2026-05-12T20:58:59.207454Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06209","last_updated":"2024-04-25T07:12:39Z","snapshot_observed_at":"2026-07-06T17:14:32.455890Z","submitted_at":"2024-01-11T18:58:36Z","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2401.06209","doi":"10.48550/arxiv.2401.06209","metadata_source":"arxiv_reference","pith_arxiv_id":"2401.06209","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Eyes wide shut? exploring the visual shortcomings of multimodal llms","venue":"arXiv (Cornell University)","work_id":"fced738d-327a-4c32-b51c-e6df45b47786","year":2024},"citing_paper":{"arxiv_id":"2404.18930","last_updated":"2025-04-01T18:36:08Z","snapshot_observed_at":"2026-08-06T19:08:14.800394Z","submitted_at":"2024-04-29T17:59:41Z","title":"Hallucination of Multimodal Large Language Models: A Survey","version":2},"reference_index":156,"source":"pdf_text","source_observed_at":"2026-05-11T12:33:32.631346Z"},"links":{"cited_paper":"/paper/2401.06209","citing_paper":"/paper/2404.18930"},"observation_digest":"sha256:f9d53c364e715b83a8889dda2dd2cbbdbd467162e9b4d95825a81117993ef65c","observation_id":"6c02f745-d645-4819-9829-623ccc82d368","resolution":{"observed_at":"2026-05-11T12:33:33.027756Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06209","last_updated":"2024-04-25T07:12:39Z","snapshot_observed_at":"2026-07-06T17:14:32.455890Z","submitted_at":"2024-01-11T18:58:36Z","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06209","snapshot_observed_at":"2026-08-07T14:45:32.061891Z","title":"Please describe this image in detail","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17812","last_updated":"2025-05-23T12:29:00Z","snapshot_observed_at":"2026-08-09T18:32:44.877470Z","submitted_at":"2025-05-23T12:29:00Z","title":"Seeing It or Not? Interpretable Vision-aware Latent Steering to Mitigate Object Hallucinations","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-07T14:45:32.061891Z"},"links":{"cited_paper":"/paper/2401.06209","citing_paper":"/paper/2505.17812"},"observation_digest":"sha256:c1e43a9407771666438d754820f179351b361db9dd981bbddd37f37c1ac52060","observation_id":"e4e60157-5b8f-4138-bf26-7e5b1f2d07bd","resolution":{"observed_at":"2026-08-07T14:45:32.061891Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06209","last_updated":"2024-04-25T07:12:39Z","snapshot_observed_at":"2026-07-06T17:14:32.455890Z","submitted_at":"2024-01-11T18:58:36Z","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06209","snapshot_observed_at":"2026-08-07T12:35:28.325476Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.24519","last_updated":"2025-05-30T12:30:50Z","snapshot_observed_at":"2026-08-09T16:33:06.604471Z","submitted_at":"2025-05-30T12:30:50Z","title":"AMIA: Automatic Masking and Joint Intention Analysis Makes LVLMs Robust Jailbreak Defenders","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-07T12:35:28.325476Z"},"links":{"cited_paper":"/paper/2401.06209","citing_paper":"/paper/2505.24519"},"observation_digest":"sha256:158a516fe5e1f47eb307909a564d152ae1cf3f6ccc9a19b557eac67b42141440","observation_id":"91e87091-4d6a-46ae-b4f2-9f348ae857a9","resolution":{"observed_at":"2026-08-07T12:35:28.325476Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06209","last_updated":"2024-04-25T07:12:39Z","snapshot_observed_at":"2026-07-06T17:14:32.455890Z","submitted_at":"2024-01-11T18:58:36Z","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06209","snapshot_observed_at":"2026-08-07T11:54:57.006847Z","title":"van den Oord, A., Vinyals, O., and Kavukcuoglu, K","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.01182","last_updated":"2025-07-08T20:18:16Z","snapshot_observed_at":"2026-08-09T08:46:08.835639Z","submitted_at":"2025-06-01T21:33:36Z","title":"Humanoid World Models: Open World Foundation Models for Humanoid Robotics","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T11:54:57.006847Z"},"links":{"cited_paper":"/paper/2401.06209","citing_paper":"/paper/2506.01182"},"observation_digest":"sha256:54551daf85343ffe9ddd1421d614cea98bae6ba709a3d4f8138059433348dbbe","observation_id":"514944c9-b9b9-427f-99fb-7d2e0502644a","resolution":{"observed_at":"2026-08-07T11:54:57.006847Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06209","last_updated":"2024-04-25T07:12:39Z","snapshot_observed_at":"2026-07-06T17:14:32.455890Z","submitted_at":"2024-01-11T18:58:36Z","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06209","snapshot_observed_at":"2026-08-07T11:39:46.653047Z","title":"Eyes wide shut? exploring the visual shortcomings of multimodal llms, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01850","last_updated":"2026-06-05T03:03:30Z","snapshot_observed_at":"2026-08-10T02:04:41.178239Z","submitted_at":"2025-06-02T16:38:50Z","title":"MoDA: Modulation Adapter for Fine-Grained Visual Grounding in Instructional MLLMs","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T11:39:46.653047Z"},"links":{"cited_paper":"/paper/2401.06209","citing_paper":"/paper/2506.01850"},"observation_digest":"sha256:09c6910a4eac300fada3dd6dd5eed3e03e2021736bd679545a1aa12811ed79a0","observation_id":"bafe4089-8d7e-40e5-aae8-f71239d64ec9","resolution":{"observed_at":"2026-08-07T11:39:46.653047Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06209","last_updated":"2024-04-25T07:12:39Z","snapshot_observed_at":"2026-07-06T17:14:32.455890Z","submitted_at":"2024-01-11T18:58:36Z","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06209","snapshot_observed_at":"2026-08-07T10:41:23.165468Z","title":"Eyes wide shut? exploring the visual shortcomings of multimodal llms.arXiv preprint arXiv:2401.06209, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.04633","last_updated":"2025-06-05T05:09:46Z","snapshot_observed_at":"2026-08-07T10:34:45.625540Z","submitted_at":"2025-06-05T05:09:46Z","title":"Unfolding Spatial Cognition: Evaluating Multimodal Models on Visual Simulations","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T10:41:23.165468Z"},"links":{"cited_paper":"/paper/2401.06209","citing_paper":"/paper/2506.04633"},"observation_digest":"sha256:913183242e9f7d1250985e9905d752dbb303d7dbe7505068f726bf25c0f8d899","observation_id":"c4d2c4f5-7971-4722-9fd1-74b1712a1143","resolution":{"observed_at":"2026-08-07T10:41:23.165468Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06209","last_updated":"2024-04-25T07:12:39Z","snapshot_observed_at":"2026-07-06T17:14:32.455890Z","submitted_at":"2024-01-11T18:58:36Z","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06209","snapshot_observed_at":"2026-08-07T05:45:56.153036Z","title":"Eyes wide shut? exploring the visual shortcomings of multimodal llms,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.07214","last_updated":"2025-06-08T16:40:40Z","snapshot_observed_at":"2026-08-09T06:16:28.205414Z","submitted_at":"2025-06-08T16:40:40Z","title":"Backdoor Attack on Vision Language Models with Stealthy Semantic Manipulation","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T05:45:56.153036Z"},"links":{"cited_paper":"/paper/2401.06209","citing_paper":"/paper/2506.07214"},"observation_digest":"sha256:67ca1fde25c6e629d8f9186556f42b97e1a032a162f45f840af348484037fd4e","observation_id":"03aea609-f8a5-465a-bd29-24afa8a79559","resolution":{"observed_at":"2026-08-07T05:45:56.153036Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06209","last_updated":"2024-04-25T07:12:39Z","snapshot_observed_at":"2026-07-06T17:14:32.455890Z","submitted_at":"2024-01-11T18:58:36Z","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06209","snapshot_observed_at":"2026-08-07T05:21:35.113327Z","title":"Eyes wide shut? exploring the visual shortcomings of multimodal llms.arXiv preprint arXiv:2401.06209, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.08227","last_updated":"2025-06-09T20:53:43Z","snapshot_observed_at":"2026-08-07T05:14:41.333172Z","submitted_at":"2025-06-09T20:53:43Z","title":"A Good CREPE needs more than just Sugar: Investigating Biases in Compositional Vision-Language Benchmarks","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T05:21:35.113327Z"},"links":{"cited_paper":"/paper/2401.06209","citing_paper":"/paper/2506.08227"},"observation_digest":"sha256:c3b4e4d949a4c2ecbd6540996494973e72d9ab58f71bfaeafec74963c8fc5735","observation_id":"a3698515-14c1-45aa-a30f-59ac2fb2f97a","resolution":{"observed_at":"2026-08-07T05:21:35.113327Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06209","last_updated":"2024-04-25T07:12:39Z","snapshot_observed_at":"2026-07-06T17:14:32.455890Z","submitted_at":"2024-01-11T18:58:36Z","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2401.06209","doi":"10.48550/arxiv.2401.06209","metadata_source":"arxiv_reference","pith_arxiv_id":"2401.06209","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Eyes wide shut? exploring the visual shortcomings of multimodal llms","venue":"arXiv (Cornell University)","work_id":"fced738d-327a-4c32-b51c-e6df45b47786","year":2024},"citing_paper":{"arxiv_id":"2507.01955","last_updated":"2026-05-01T00:57:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-02T17:59:07Z","title":"How Well Does GPT-4o Understand Vision? Evaluating Multimodal Foundation Models on Standard Computer Vision Tasks","version":3},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-05-19T05:55:09.188048Z"},"links":{"cited_paper":"/paper/2401.06209","citing_paper":"/paper/2507.01955"},"observation_digest":"sha256:a540c3be74bf78349e5d475346af5e88aa8907f2f430a14e319c536f30177327","observation_id":"ffedd751-f1f1-40a0-92bd-419d0354ae2e","resolution":{"observed_at":"2026-05-19T05:57:08.023724Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06209","last_updated":"2024-04-25T07:12:39Z","snapshot_observed_at":"2026-07-06T17:14:32.455890Z","submitted_at":"2024-01-11T18:58:36Z","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06209","snapshot_observed_at":"2026-08-05T13:24:39.817974Z","title":"Eyes wide shut? exploring the visual shortcomings of multimodal llms, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.00676","last_updated":"2025-08-31T03:08:02Z","snapshot_observed_at":"2026-08-07T21:36:21.058533Z","submitted_at":"2025-08-31T03:08:02Z","title":"LLaVA-Critic-R1: Your Critic Model is Secretly a Strong Policy Model","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-05T13:24:39.817974Z"},"links":{"cited_paper":"/paper/2401.06209","citing_paper":"/paper/2509.00676"},"observation_digest":"sha256:d5ec72834e6e33b3b5907f97de60c3ed944e3a365e0ea25e38d14c46741a2f9b","observation_id":"fa75da25-bfe4-4e13-822e-b46c35599431","resolution":{"observed_at":"2026-08-05T13:24:39.817974Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06209","last_updated":"2024-04-25T07:12:39Z","snapshot_observed_at":"2026-07-06T17:14:32.455890Z","submitted_at":"2024-01-11T18:58:36Z","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06209","snapshot_observed_at":"2026-08-05T11:56:10.921205Z","title":"Eyes wide shut? exploring the visual shortcomings of multimodal llms, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.02175","last_updated":"2025-09-04T16:38:44Z","snapshot_observed_at":"2026-08-05T11:56:06.572610Z","submitted_at":"2025-09-02T10:32:58Z","title":"Understanding Space Is Rocket Science -- Only Top Reasoning Models Can Solve Spatial Understanding Tasks","version":2},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-05T11:56:10.921205Z"},"links":{"cited_paper":"/paper/2401.06209","citing_paper":"/paper/2509.02175"},"observation_digest":"sha256:6f641f2ca68bcd61b23e3882e8f50c7c015d0cd5e392f559cbc41c11975ed0ec","observation_id":"1480ea88-014d-41e3-91a9-1e552c22266b","resolution":{"observed_at":"2026-08-05T11:56:10.921205Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06209","last_updated":"2024-04-25T07:12:39Z","snapshot_observed_at":"2026-07-06T17:14:32.455890Z","submitted_at":"2024-01-11T18:58:36Z","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2401.06209","doi":"10.48550/arxiv.2401.06209","metadata_source":"arxiv_reference","pith_arxiv_id":"2401.06209","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Eyes wide shut? exploring the visual shortcomings of multimodal llms","venue":"arXiv (Cornell University)","work_id":"fced738d-327a-4c32-b51c-e6df45b47786","year":2024},"citing_paper":{"arxiv_id":"2511.21678","last_updated":"2026-05-02T09:08:51Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-11-26T18:55:08Z","title":"Agentic Learner with Grow-and-Refine Multimodal Semantic Memory","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-17T04:27:40.232015Z"},"links":{"cited_paper":"/paper/2401.06209","citing_paper":"/paper/2511.21678"},"observation_digest":"sha256:6df5ca2f8e6637c018eedc5053f8e37c351f477a9dbb2e59721efbded8f96342","observation_id":"dd397fc2-4ae1-42b8-ba90-65c061c5ff18","resolution":{"observed_at":"2026-05-17T04:29:01.594708Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06209","last_updated":"2024-04-25T07:12:39Z","snapshot_observed_at":"2026-07-06T17:14:32.455890Z","submitted_at":"2024-01-11T18:58:36Z","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2401.06209","doi":"10.48550/arxiv.2401.06209","metadata_source":"arxiv_reference","pith_arxiv_id":"2401.06209","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Eyes wide shut? exploring the visual shortcomings of multimodal llms","venue":"arXiv (Cornell University)","work_id":"fced738d-327a-4c32-b51c-e6df45b47786","year":2024},"citing_paper":{"arxiv_id":"2602.02276","last_updated":"2026-02-02T16:17:38Z","snapshot_observed_at":"2026-07-06T22:44:09.804048Z","submitted_at":"2026-02-02T16:17:38Z","title":"Kimi K2.5: Visual Agentic Intelligence","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-10T16:09:05.225767Z"},"links":{"cited_paper":"/paper/2401.06209","citing_paper":"/paper/2602.02276"},"observation_digest":"sha256:70f4ca4891f87eac52918acf19399e06712ba37a6ca33b298288ac076f082745","observation_id":"61285ced-81ef-45f4-aa01-93533de21893","resolution":{"observed_at":"2026-05-10T16:09:05.382165Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06209","last_updated":"2024-04-25T07:12:39Z","snapshot_observed_at":"2026-07-06T17:14:32.455890Z","submitted_at":"2024-01-11T18:58:36Z","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2401.06209","doi":"10.48550/arxiv.2401.06209","metadata_source":"arxiv_reference","pith_arxiv_id":"2401.06209","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Eyes wide shut? exploring the visual shortcomings of multimodal llms","venue":"arXiv (Cornell University)","work_id":"fced738d-327a-4c32-b51c-e6df45b47786","year":2024},"citing_paper":{"arxiv_id":"2602.13310","last_updated":"2026-05-07T08:00:36Z","snapshot_observed_at":"2026-08-03T02:11:51.693223Z","submitted_at":"2026-02-10T03:53:25Z","title":"Visual Para-Thinker: Divide-and-Conquer Reasoning for Visual Comprehension","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-16T03:27:53.694506Z"},"links":{"cited_paper":"/paper/2401.06209","citing_paper":"/paper/2602.13310"},"observation_digest":"sha256:4f4c850d80394fb24c48a8522a5ca8d2e82f716b0812fa6deb28bfc405879a86","observation_id":"d7fc9b8a-1cc8-4058-8cb7-618ac66d9d06","resolution":{"observed_at":"2026-05-16T03:30:33.014248Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06209","last_updated":"2024-04-25T07:12:39Z","snapshot_observed_at":"2026-07-06T17:14:32.455890Z","submitted_at":"2024-01-11T18:58:36Z","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06209","snapshot_observed_at":"2026-07-14T22:19:38.973091Z","title":"Eyes wide shut? exploring the visual shortcomings of multimodal llms.arXiv preprint arXiv:2401.06209, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.12433","last_updated":"2026-06-02T23:51:35Z","snapshot_observed_at":"2026-08-06T21:04:01.499410Z","submitted_at":"2026-03-12T20:33:19Z","title":"Revisiting Model Stitching In the Foundation Model Era","version":3},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-07-14T22:19:38.973091Z"},"links":{"cited_paper":"/paper/2401.06209","citing_paper":"/paper/2603.12433"},"observation_digest":"sha256:8a975870f5bc83da4b7dce97ebf0c5c8541adc9fe4f54259a9e095a1fe3683ed","observation_id":"8d35bc86-3a3c-4cfc-be9f-1b1a390b3d28","resolution":{"observed_at":"2026-07-14T22:19:38.973091Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06209","last_updated":"2024-04-25T07:12:39Z","snapshot_observed_at":"2026-07-06T17:14:32.455890Z","submitted_at":"2024-01-11T18:58:36Z","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2401.06209","doi":"10.48550/arxiv.2401.06209","metadata_source":"arxiv_reference","pith_arxiv_id":"2401.06209","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Eyes wide shut? exploring the visual shortcomings of multimodal llms","venue":"arXiv (Cornell University)","work_id":"fced738d-327a-4c32-b51c-e6df45b47786","year":2024},"citing_paper":{"arxiv_id":"2604.10219","last_updated":"2026-05-28T16:20:33Z","snapshot_observed_at":"2026-08-01T16:49:58.717615Z","submitted_at":"2026-04-11T13:59:05Z","title":"Cognitive Pivot Points and Visual Anchoring: Unveiling and Rectifying Hallucinations in Multimodal Reasoning Models","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-05-10T16:03:15.222571Z"},"links":{"cited_paper":"/paper/2401.06209","citing_paper":"/paper/2604.10219"},"observation_digest":"sha256:fe878e538d8eea4b05f1e57bb549f382fdab1dfd5bb11e2d81f496afbc37cdaf","observation_id":"b0d5d9e9-0238-4fb9-bc1b-1f2f903be8a2","resolution":{"observed_at":"2026-05-11T09:25:59.280081Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06209","last_updated":"2024-04-25T07:12:39Z","snapshot_observed_at":"2026-07-06T17:14:32.455890Z","submitted_at":"2024-01-11T18:58:36Z","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06209","snapshot_observed_at":"2026-07-12T22:48:45.647588Z","title":"Available: https://arxiv.org/abs/2401.06209","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2604.10219","last_updated":"2026-05-28T16:20:33Z","snapshot_observed_at":"2026-08-01T16:49:58.717615Z","submitted_at":"2026-04-11T13:59:05Z","title":"Cognitive Pivot Points and Visual Anchoring: Unveiling and Rectifying Hallucinations in Multimodal Reasoning Models","version":2},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-07-12T22:48:45.647588Z"},"links":{"cited_paper":"/paper/2401.06209","citing_paper":"/paper/2604.10219"},"observation_digest":"sha256:6f8da8fa73a3481a2ccdc126382ecffd78e693dd578e77a3c92e6920d9c2e454","observation_id":"88b072a1-0dd7-4ab6-9610-fcb7454daba0","resolution":{"observed_at":"2026-07-12T22:48:45.647588Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06209","last_updated":"2024-04-25T07:12:39Z","snapshot_observed_at":"2026-07-06T17:14:32.455890Z","submitted_at":"2024-01-11T18:58:36Z","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2401.06209","doi":"10.48550/arxiv.2401.06209","metadata_source":"arxiv_reference","pith_arxiv_id":"2401.06209","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Eyes wide shut? exploring the visual shortcomings of multimodal llms","venue":"arXiv (Cornell University)","work_id":"fced738d-327a-4c32-b51c-e6df45b47786","year":2024},"citing_paper":{"arxiv_id":"2604.13803","last_updated":"2026-04-15T12:38:51Z","snapshot_observed_at":"2026-07-06T23:01:41.643335Z","submitted_at":"2026-04-15T12:38:51Z","title":"Gaslight, Gatekeep, V1-V3: Early Visual Cortex Alignment Shields Vision-Language Models from Sycophantic Manipulation","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-05-10T13:09:35.407790Z"},"links":{"cited_paper":"/paper/2401.06209","citing_paper":"/paper/2604.13803"},"observation_digest":"sha256:1ddd3388866658ffa2bdb18c4daddce044c6433f874cb7abaca5cad1a9483969","observation_id":"52b91b38-6d68-4b2f-a4a7-ca77ac00eb1f","resolution":{"observed_at":"2026-05-10T13:10:26.177537Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06209","last_updated":"2024-04-25T07:12:39Z","snapshot_observed_at":"2026-07-06T17:14:32.455890Z","submitted_at":"2024-01-11T18:58:36Z","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2401.06209","doi":"10.48550/arxiv.2401.06209","metadata_source":"arxiv_reference","pith_arxiv_id":"2401.06209","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Eyes wide shut? exploring the visual shortcomings of multimodal llms","venue":"arXiv (Cornell University)","work_id":"fced738d-327a-4c32-b51c-e6df45b47786","year":2024},"citing_paper":{"arxiv_id":"2604.16883","last_updated":"2026-04-18T07:23:22Z","snapshot_observed_at":"2026-08-03T01:53:31.909118Z","submitted_at":"2026-04-18T07:23:22Z","title":"SinkRouter: Sink-Aware Routing for Efficient Long-Context Decoding in Large Language and Multimodal Models","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-10T07:56:05.583390Z"},"links":{"cited_paper":"/paper/2401.06209","citing_paper":"/paper/2604.16883"},"observation_digest":"sha256:dda21d3db60c76b959c3b3787d5c7067f34a3bc2706b783b825310c6eb6a9273","observation_id":"1cea5ba8-9582-406c-91f6-83972d727325","resolution":{"observed_at":"2026-05-10T07:57:15.441339Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06209","last_updated":"2024-04-25T07:12:39Z","snapshot_observed_at":"2026-07-06T17:14:32.455890Z","submitted_at":"2024-01-11T18:58:36Z","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2401.06209","doi":"10.48550/arxiv.2401.06209","metadata_source":"arxiv_reference","pith_arxiv_id":"2401.06209","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Eyes wide shut? exploring the visual shortcomings of multimodal llms","venue":"arXiv (Cornell University)","work_id":"fced738d-327a-4c32-b51c-e6df45b47786","year":2024},"citing_paper":{"arxiv_id":"2604.20665","last_updated":"2026-05-21T08:47:56Z","snapshot_observed_at":"2026-08-06T16:05:08.769769Z","submitted_at":"2026-04-22T15:15:32Z","title":"The Expense of Seeing: Attaining Trustworthy Multimodal Reasoning Within the Monolithic Paradigm","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-09T23:56:02.856878Z"},"links":{"cited_paper":"/paper/2401.06209","citing_paper":"/paper/2604.20665"},"observation_digest":"sha256:45a9a2dcc076980b29ca1d41d647519b0db7b2d5147eb57b153cedd47e97c0fb","observation_id":"b574e902-cc1f-4634-8f37-eccd69075fc3","resolution":{"observed_at":"2026-05-11T13:51:05.103821Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06209","last_updated":"2024-04-25T07:12:39Z","snapshot_observed_at":"2026-07-06T17:14:32.455890Z","submitted_at":"2024-01-11T18:58:36Z","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2401.06209","doi":"10.48550/arxiv.2401.06209","metadata_source":"arxiv_reference","pith_arxiv_id":"2401.06209","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Eyes wide shut? exploring the visual shortcomings of multimodal llms","venue":"arXiv (Cornell University)","work_id":"fced738d-327a-4c32-b51c-e6df45b47786","year":2024},"citing_paper":{"arxiv_id":"2604.20665","last_updated":"2026-05-21T08:47:56Z","snapshot_observed_at":"2026-08-06T16:05:08.769769Z","submitted_at":"2026-04-22T15:15:32Z","title":"The Expense of Seeing: Attaining Trustworthy Multimodal Reasoning Within the Monolithic Paradigm","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-22T10:35:50.838150Z"},"links":{"cited_paper":"/paper/2401.06209","citing_paper":"/paper/2604.20665"},"observation_digest":"sha256:dc7bd9f79928f425187930c24c5ea5c7b288ee9cf4d06fa274f73e503e7dd855","observation_id":"5bf38efe-c0b3-4175-ae13-d0f1cb80751c","resolution":{"observed_at":"2026-05-22T10:36:24.988618Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06209","last_updated":"2024-04-25T07:12:39Z","snapshot_observed_at":"2026-07-06T17:14:32.455890Z","submitted_at":"2024-01-11T18:58:36Z","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2401.06209","doi":"10.48550/arxiv.2401.06209","metadata_source":"arxiv_reference","pith_arxiv_id":"2401.06209","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Eyes wide shut? exploring the visual shortcomings of multimodal llms","venue":"arXiv (Cornell University)","work_id":"fced738d-327a-4c32-b51c-e6df45b47786","year":2024},"citing_paper":{"arxiv_id":"2604.21027","last_updated":"2026-08-02T16:15:42Z","snapshot_observed_at":"2026-08-06T23:24:26.820832Z","submitted_at":"2026-04-22T19:18:36Z","title":"HypEHR: Hyperbolic Modeling of Electronic Health Records for Efficient Question Answering","version":1},"reference_index":251,"source":"arxiv_source","source_observed_at":"2026-05-09T23:51:47.724033Z"},"links":{"cited_paper":"/paper/2401.06209","citing_paper":"/paper/2604.21027"},"observation_digest":"sha256:103e6ad20034548537c455d6df21338769d860f5ddbe70f7e8d79b7dde936486","observation_id":"4ad5867c-504a-44f8-be22-bb2dda432282","resolution":{"observed_at":"2026-05-09T23:54:45.492581Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06209","last_updated":"2024-04-25T07:12:39Z","snapshot_observed_at":"2026-07-06T17:14:32.455890Z","submitted_at":"2024-01-11T18:58:36Z","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2401.06209","doi":"10.48550/arxiv.2401.06209","metadata_source":"arxiv_reference","pith_arxiv_id":"2401.06209","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Eyes wide shut? exploring the visual shortcomings of multimodal llms","venue":"arXiv (Cornell University)","work_id":"fced738d-327a-4c32-b51c-e6df45b47786","year":2024},"citing_paper":{"arxiv_id":"2605.26380","last_updated":"2026-05-25T23:01:05Z","snapshot_observed_at":"2026-08-09T05:28:34.155888Z","submitted_at":"2026-05-25T23:01:05Z","title":"VisualNeedle: Benchmarking Active Visual Search in Information-Dense Scenes","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-29T22:14:00.535472Z"},"links":{"cited_paper":"/paper/2401.06209","citing_paper":"/paper/2605.26380"},"observation_digest":"sha256:6d7f25525b510f509b4b80bdfc05d852db6eb66e7231aff722d2ea193b0829e2","observation_id":"e12fcc5e-a825-45fb-be72-40741c6758aa","resolution":{"observed_at":"2026-06-29T22:23:59.757075Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06209","last_updated":"2024-04-25T07:12:39Z","snapshot_observed_at":"2026-07-06T17:14:32.455890Z","submitted_at":"2024-01-11T18:58:36Z","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2401.06209","doi":"10.48550/arxiv.2401.06209","metadata_source":"arxiv_reference","pith_arxiv_id":"2401.06209","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Eyes wide shut? exploring the visual shortcomings of multimodal llms","venue":"arXiv (Cornell University)","work_id":"fced738d-327a-4c32-b51c-e6df45b47786","year":2024},"citing_paper":{"arxiv_id":"2606.00384","last_updated":"2026-06-07T19:48:24Z","snapshot_observed_at":"2026-08-08T23:35:17.895270Z","submitted_at":"2026-05-29T21:57:37Z","title":"VESTA: Visual Exploration with Statistical Tool Agents","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-06-28T21:58:11.339217Z"},"links":{"cited_paper":"/paper/2401.06209","citing_paper":"/paper/2606.00384"},"observation_digest":"sha256:3fb99b677145db8a334f5adbf0e24c37d4f9e0f1c5244e6f576584b1e977dd26","observation_id":"589070f9-af55-4bd1-94c9-a01ecd1e9b50","resolution":{"observed_at":"2026-07-01T19:56:10.468376Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06209","last_updated":"2024-04-25T07:12:39Z","snapshot_observed_at":"2026-07-06T17:14:32.455890Z","submitted_at":"2024-01-11T18:58:36Z","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2401.06209","doi":"10.48550/arxiv.2401.06209","metadata_source":"arxiv_reference","pith_arxiv_id":"2401.06209","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Eyes wide shut? exploring the visual shortcomings of multimodal llms","venue":"arXiv (Cornell University)","work_id":"fced738d-327a-4c32-b51c-e6df45b47786","year":2024},"citing_paper":{"arxiv_id":"2606.11854","last_updated":"2026-06-10T09:30:37Z","snapshot_observed_at":"2026-08-01T08:16:13.928240Z","submitted_at":"2026-06-10T09:30:37Z","title":"Fine-tuning Multi-modal LLMs with ART: Art-based Reinforcement Training","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-06-27T10:30:51.490047Z"},"links":{"cited_paper":"/paper/2401.06209","citing_paper":"/paper/2606.11854"},"observation_digest":"sha256:e164013415d11000729e9b67033036c2ea6ba1cc5b98cef1f9380ead92261ed3","observation_id":"0d379c7f-0032-4471-934d-5751010ed5f5","resolution":{"observed_at":"2026-07-03T09:07:48.143928Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06209","last_updated":"2024-04-25T07:12:39Z","snapshot_observed_at":"2026-07-06T17:14:32.455890Z","submitted_at":"2024-01-11T18:58:36Z","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2401.06209","doi":"10.48550/arxiv.2401.06209","metadata_source":"arxiv_reference","pith_arxiv_id":"2401.06209","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Eyes wide shut? exploring the visual shortcomings of multimodal llms","venue":"arXiv (Cornell University)","work_id":"fced738d-327a-4c32-b51c-e6df45b47786","year":2024},"citing_paper":{"arxiv_id":"2606.16009","last_updated":"2026-06-16T12:53:59Z","snapshot_observed_at":"2026-07-06T23:52:37.922109Z","submitted_at":"2026-06-14T20:41:59Z","title":"Bridging the Usability Gap: Lessons from Interpreting Studies for Machine Interpreting Design","version":2},"reference_index":87,"source":"arxiv_source","source_observed_at":"2026-06-27T03:34:37.373928Z"},"links":{"cited_paper":"/paper/2401.06209","citing_paper":"/paper/2606.16009"},"observation_digest":"sha256:4b08f5195cae8efe1f2e669c475d1316922454980ded046acfd280a087f227a6","observation_id":"72f3f772-e9ff-400a-af7d-525651eb08f5","resolution":{"observed_at":"2026-06-27T03:40:27.972918Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06209","last_updated":"2024-04-25T07:12:39Z","snapshot_observed_at":"2026-07-06T17:14:32.455890Z","submitted_at":"2024-01-11T18:58:36Z","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2401.06209","doi":"10.48550/arxiv.2401.06209","metadata_source":"arxiv_reference","pith_arxiv_id":"2401.06209","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Eyes wide shut? exploring the visual shortcomings of multimodal llms","venue":"arXiv (Cornell University)","work_id":"fced738d-327a-4c32-b51c-e6df45b47786","year":2024},"citing_paper":{"arxiv_id":"2606.18974","last_updated":"2026-06-25T02:14:41Z","snapshot_observed_at":"2026-08-02T20:17:43.258482Z","submitted_at":"2026-06-17T11:59:15Z","title":"Visual-OPSD: Cross-Modal On-Policy Self-Distillation for Efficient Unified Multimodal Reasoning","version":2},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-06-26T21:47:51.437284Z"},"links":{"cited_paper":"/paper/2401.06209","citing_paper":"/paper/2606.18974"},"observation_digest":"sha256:00f3520f8f23fc04920e6483b9863f5b84ee223c020db3b43b6ed8c87dadd65c","observation_id":"79206ccc-5d08-40bb-bc08-6dacc2b386d8","resolution":{"observed_at":"2026-07-03T23:49:02.268563Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06209","last_updated":"2024-04-25T07:12:39Z","snapshot_observed_at":"2026-07-06T17:14:32.455890Z","submitted_at":"2024-01-11T18:58:36Z","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06209","snapshot_observed_at":"2026-07-12T11:31:14.532101Z","title":"Eyes wide shut? exploring the visual shortcomings of multimodal llms, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.02551","last_updated":"2026-06-26T16:05:57Z","snapshot_observed_at":"2026-08-06T08:59:01.715141Z","submitted_at":"2026-06-26T16:05:57Z","title":"DELTAVID: Enhancing Fine-Grained Spatiotemporal Perception with Cross-Video Differences","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-07-12T11:31:14.532101Z"},"links":{"cited_paper":"/paper/2401.06209","citing_paper":"/paper/2607.02551"},"observation_digest":"sha256:350a09eb30474def985451023757b00c6f1ee318074b180ec5488db15cbd7629","observation_id":"1c4639ee-9939-43cf-8533-2a3508e43a68","resolution":{"observed_at":"2026-07-12T11:31:14.532101Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06209","last_updated":"2024-04-25T07:12:39Z","snapshot_observed_at":"2026-07-06T17:14:32.455890Z","submitted_at":"2024-01-11T18:58:36Z","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06209","snapshot_observed_at":"2026-07-31T02:56:58.277821Z","title":"Eyes wide shut? exploring the visual shortcomings of multimodal llms, 2024 b","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.28590","last_updated":"2026-07-30T17:43:46Z","snapshot_observed_at":"2026-08-06T14:14:02.472307Z","submitted_at":"2026-07-30T17:43:46Z","title":"VAD: Attributing Visual Evidence for Target Reconstruction in Multimodal On-Policy Distillation","version":1},"reference_index":88,"source":"arxiv_source","source_observed_at":"2026-07-31T02:56:58.277821Z"},"links":{"cited_paper":"/paper/2401.06209","citing_paper":"/paper/2607.28590"},"observation_digest":"sha256:26c5acfd6156ce6e34aee0699024ed03d457ac6bffd5c9eb3333a7bac4aa1c85","observation_id":"9c1a1b0b-26a0-4323-8fdc-0a01a5581cfa","resolution":{"observed_at":"2026-07-31T02:56:58.277821Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06209","last_updated":"2024-04-25T07:12:39Z","snapshot_observed_at":"2026-07-06T17:14:32.455890Z","submitted_at":"2024-01-11T18:58:36Z","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06209","snapshot_observed_at":"2026-08-07T04:24:54.770078Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.06361","last_updated":"2026-08-06T17:57:06Z","snapshot_observed_at":"2026-08-09T23:13:31.191909Z","submitted_at":"2026-08-06T17:57:06Z","title":"The Low Frequency Trap: Video Language Models Fail at Simple Event Bookkeeping","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-07T04:24:54.770078Z"},"links":{"cited_paper":"/paper/2401.06209","citing_paper":"/paper/2608.06361"},"observation_digest":"sha256:f900dc0a97d24b5085b34a0def3f09a8befd28dd66fb06e94874ed2085de8741","observation_id":"8e51ee67-2932-4e6c-a1ff-4b3e0843eb3e","resolution":{"observed_at":"2026-08-07T04:24:54.770078Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2401.06209/citation-record","integrity":"/paper/2401.06209/integrity","json":"/paper/2401.06209/citation-record.json","paper":"/paper/2401.06209"},"outbound":[],"paper":{"arxiv_id":"2401.06209","last_updated":"2024-04-25T07:12:39Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T17:14:32.455890Z","submitted_at":"2024-01-11T18:58:36Z","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 33 inbound Pith citation observations for arXiv:2401.06209."}