{"as_of":"2026-08-19T19:50:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:5c7df26017f2a1692614cc7f8d8b7e5ec440a9282c7a6e25bed3c40b90755b34","coverage":[{"denominator":67,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":67,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T22:37:17.470659Z","state":"measured"},{"denominator":67,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":67,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-19T06:32:44.657259+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2505.06840/citation-record","integrity":"/paper/2505.06840/integrity","json":"/paper/2505.06840/citation-record.json","paper":"/paper/2505.06840"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:18.598009Z","title":"https://cdn.openai.com/papers/GPTV_System_Card.pdf, 2023","venue":null,"work_id":"ab60ffda-7d16-4fe1-9b5f-65df1c170d8c","year":2023},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.155881Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:36ab89c57ceddfe61aa372b9ebcda6b06ffc7f6817e7db2e767f6972c073b174","observation_id":"0a287be2-67bd-4035-b723-c1121d674098","resolution":{"observed_at":"2026-08-15T22:37:18.602767Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:18.581759Z","title":"https://sharegpt.com, 2023","venue":null,"work_id":"5e4a7c94-ed67-4651-96d7-f62c779df1aa","year":2023},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.161282Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:76b1773f08af3a45ab9259495e6a63e192e75cc09e88345707e2d0ce5f9a6578","observation_id":"4ac889ba-2a7b-4497-969e-7d02ebbdda3e","resolution":{"observed_at":"2026-08-15T22:37:18.587026Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:18.564089Z","title":"https://www-cdn.anthropic.com/ de8ba9b01c9ab7cbabf5c33b80b7bbc618857627/Model_Card_Claude_3.pdf, 2024","venue":null,"work_id":"406143fa-5fa3-477b-8284-cc0930230f07","year":2024},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.166532Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:848da700176befb0b2be8043cf7389bb7c2cbd40ffff302eec4bbac58a48dd4f","observation_id":"9d324040-7351-40d1-8e58-4d8d6bbb40b2","resolution":{"observed_at":"2026-08-15T22:37:18.569607Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:18.547828Z","title":"https://huggingface.co/datasets/laion/gpt4v-dataset, 2024","venue":null,"work_id":"ab0d0d27-9672-4c49-ae0e-2b84ac7ce50b","year":2024},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.171579Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:24b5559c95eaa4894865acc19f49a2fc9a66b031105b9ca6eab12339cebd384d","observation_id":"4fc095c5-bcce-4362-ac18-254801f0d5d3","resolution":{"observed_at":"2026-08-15T22:37:18.552767Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:18.532185Z","title":"https://huggingface.co/NousResearch/Nous-Hermes-2-Yi-34B , 2024","venue":null,"work_id":"2877b758-bd04-496e-bc48-aa3ee5bad711","year":2024},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.176151Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:135e3dfcfc6dd74e4ebe9cb96cf323327502195f58ff45e38ed49a7d87406a79","observation_id":"a5e378aa-f1df-44ce-bbd6-1d0f8620ec8b","resolution":{"observed_at":"2026-08-15T22:37:18.537366Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:18.514920Z","title":"Flamingo: a visual language model for few-shot learning","venue":null,"work_id":"d0ce3a92-7991-4535-9139-bc03dbd55cbc","year":2022},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.180913Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:2140824839d77e11ed3b40ff67703956b315f240aae4f00191a1f5e19e295485","observation_id":"b571747b-821f-4970-ab66-a7873f24cd97","resolution":{"observed_at":"2026-08-15T22:37:18.521160Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:18.496558Z","title":"Lawrence Zitnick, and Devi Parikh","venue":null,"work_id":"3914625c-48d1-4cf4-8b7d-7782f73be280","year":2015},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.185939Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:e2373ef1977eedc5391c63db4fa69161335c42171b1f1376dcafe2f777d8e5a9","observation_id":"989e12f3-2a98-4604-98f0-572d2c14524f","resolution":{"observed_at":"2026-08-15T22:37:18.502876Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:18.478569Z","title":"Openflamingo, Mar","venue":null,"work_id":"c35fc331-8ba9-414d-bd65-cea191d79785","year":2023},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.190532Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:7b9e408d4f53ac03262f11568338dcd8e6cf20841df4e937fba9ce1da7a02a0b","observation_id":"3ac63604-86a3-4018-8483-697a7a9175e6","resolution":{"observed_at":"2026-08-15T22:37:18.484249Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.12966","last_updated":"2023-10-13T02:41:28Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-24T17:59:17Z","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.12966","snapshot_observed_at":"2026-08-15T22:37:17.194841Z","title":"Qwen-vl: A versatile vision-language model for understanding, localization, text reading, and beyond","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.194841Z"},"links":{"cited_paper":"/paper/2308.12966","citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:109aadcbd0715cfd94ef2e899121be6c132aceebe2db4fc26ab170438d10f6c2","observation_id":"e3ffb5c8-188d-411c-9bda-28c431d5f270","resolution":{"observed_at":"2026-08-15T22:37:17.194841Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:18.458551Z","title":"Language models are few-shot learners","venue":null,"work_id":"b239f339-471b-47be-91eb-133d0a171948","year":2020},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.199706Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:c7be2820e5ac98700ea569b919d4becdd3867f3585fbed375529017d2296cf9f","observation_id":"5de9a258-b435-4f85-b6b1-ddae3a41fcf0","resolution":{"observed_at":"2026-08-15T22:37:18.463902Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.11684","last_updated":"2024-06-17T07:55:59Z","snapshot_observed_at":"2026-07-06T17:31:53.996417Z","submitted_at":"2024-02-18T19:26:49Z","title":"ALLaVA: Harnessing GPT4V-Synthesized Data for Lite Vision-Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.11684","snapshot_observed_at":"2026-08-15T22:37:17.204925Z","title":"Allava: Harnessing gpt4v-synthesized data for a lite vision-language model","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.204925Z"},"links":{"cited_paper":"/paper/2402.11684","citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:e4195324815d9f2c4ab59d00ed974778862fd239d8b402ad32e82030909b9e79","observation_id":"0cbeef1e-5d27-4630-a553-675167a3ac28","resolution":{"observed_at":"2026-08-15T22:37:17.204925Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.15195","last_updated":"2023-07-03T16:08:00Z","snapshot_observed_at":"2026-08-13T14:42:26.982679Z","submitted_at":"2023-06-27T04:31:52Z","title":"Shikra: Unleashing Multimodal LLM's Referential Dialogue Magic","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.15195","snapshot_observed_at":"2026-08-15T22:37:17.210203Z","title":"Shikra: Unleashing multimodal llm’s referential dialogue magic","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.210203Z"},"links":{"cited_paper":"/paper/2306.15195","citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:c275099b956276cd22b5e4dd7ba25c28a6f1deaefab1efc2c6f2a71e98de2e8a","observation_id":"9d92596d-9fc0-4e35-bfde-20de194ac905","resolution":{"observed_at":"2026-08-15T22:37:17.210203Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.12793","last_updated":"2023-11-28T08:52:50Z","snapshot_observed_at":"2026-08-14T06:42:43.375489Z","submitted_at":"2023-11-21T18:58:11Z","title":"ShareGPT4V: Improving Large Multi-Modal Models with Better Captions","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.12793","snapshot_observed_at":"2026-08-15T22:37:17.215019Z","title":"Sharegpt4v: Improving large multi-modal models with better captions","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.215019Z"},"links":{"cited_paper":"/paper/2311.12793","citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:0b6e2e436b8dbc820164de64f8743d997b289845f2dcc5be1b0437362807ee68","observation_id":"25120bb8-baa4-4bee-a1da-a4a6a8a00c24","resolution":{"observed_at":"2026-08-15T22:37:17.215019Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.03149","last_updated":"2024-06-19T03:29:41Z","snapshot_observed_at":"2026-08-16T14:29:28.409344Z","submitted_at":"2024-01-06T07:54:58Z","title":"CaMML: Context-Aware Multimodal Learner for Large Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.03149","snapshot_observed_at":"2026-08-15T22:37:17.220422Z","title":"Camml: Context-aware multimodal learner for large models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.220422Z"},"links":{"cited_paper":"/paper/2401.03149","citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:341e6c1adbb405a112f1a91720929d6d210e4d4c4d85e18e03bb484d98385a59","observation_id":"84cf5b1f-9838-4111-99fa-1eebf1bff84b","resolution":{"observed_at":"2026-08-15T22:37:17.220422Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:18.441758Z","title":"How far are we to gpt-4v? closing the gap to commercial multimodal models with open-source suites, 2024","venue":null,"work_id":"5f00ed65-dd1f-4b67-b4e6-6b8c7637e56f","year":2024},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.225726Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:ff5f4ac93ed9a3a97df15760cb268faf9b1785a2ec1468fb84c4f57eca71cf6e","observation_id":"e435d66c-1fdb-453d-83dd-ea23f37ff8fd","resolution":{"observed_at":"2026-08-15T22:37:18.447148Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03766","last_updated":"2024-02-06T07:16:36Z","snapshot_observed_at":"2026-08-09T19:28:34.281681Z","submitted_at":"2024-02-06T07:16:36Z","title":"MobileVLM V2: Faster and Stronger Baseline for Vision Language Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.03766","snapshot_observed_at":"2026-08-15T22:37:17.230104Z","title":"Mobilevlm v2: Faster and stronger baseline for vision language model","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.230104Z"},"links":{"cited_paper":"/paper/2402.03766","citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:cd40c9291e6a4b1e4fa850f8dca5619159ab206fdc1cd7ce55e58b61b03ec298","observation_id":"330339b1-54f0-4fab-b225-c83bfe411f54","resolution":{"observed_at":"2026-08-15T22:37:17.230104Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:18.425272Z","title":"Instructblip: Towards general-purpose vision-language models with instruction tuning, 2023","venue":null,"work_id":"78314ef2-8c09-4878-b460-5ed8d1131098","year":2023},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.235144Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:f293327b95887258dd0a2edd0204157ddb24c666169a3db395c27d309b0ab88d","observation_id":"da9f4376-e813-40b7-9245-661f3807079b","resolution":{"observed_at":"2026-08-15T22:37:18.430480Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:18.408681Z","title":"Internlm-xcomposer2-4khd: A pioneering large vision-language model handling resolutions from 336 pixels to 4k hd, 2024","venue":null,"work_id":"75a9ce15-4b74-4f42-b794-bdc890a55a24","year":2024},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.239664Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:d2a2369f3aa3b4b66ffd4ac42989f2d68144eaa44cbcd8c5ec4a0eb2bf57fa53","observation_id":"7cd13423-a335-4eb1-b13c-3259b29aedfe","resolution":{"observed_at":"2026-08-15T22:37:18.414109Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:17.244378Z","title":"An image is worth 16x16 words: Transformers for image recognition at scale","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.244378Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:16e78f438b009fed31d9c6081b23e0714badda0ced6fe024a67ee11c886c894f","observation_id":"4640db6b-000f-415c-8d9b-7c2690237bf0","resolution":{"observed_at":"2026-08-15T22:37:17.244378Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.13394","last_updated":"2025-10-24T02:45:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-23T09:22:36Z","title":"MME: A Comprehensive Evaluation Benchmark for Multimodal Large Language Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.13394","snapshot_observed_at":"2026-08-15T22:37:17.248980Z","title":"Mme: A comprehensive evaluation benchmark for multimodal large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.248980Z"},"links":{"cited_paper":"/paper/2306.13394","citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:ef7b2dd76bb9476a79b65d1cf0ebf118818977a843a295a9ad3b2b238d4cd1e0","observation_id":"1ef0dc13-1c7a-46c4-806e-0624221e56bf","resolution":{"observed_at":"2026-08-15T22:37:17.248980Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:18.380929Z","title":"Vizwiz grand challenge: Answering visual questions from blind people","venue":null,"work_id":"34973f3f-ba7f-4d97-8b69-65e62457d8d2","year":2018},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.253593Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:1ef7fe701b78f53729be4a1e1bef869d10d3ed2eb86b97d39656d6b83b084299","observation_id":"3a21c9b7-cfd3-4fec-80d0-d08d58f7c302","resolution":{"observed_at":"2026-08-15T22:37:18.385984Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.08914","last_updated":"2024-12-27T06:56:18Z","snapshot_observed_at":"2026-08-19T15:30:58.990292Z","submitted_at":"2023-12-14T13:20:57Z","title":"CogAgent: A Visual Language Model for GUI Agents","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.08914","snapshot_observed_at":"2026-08-15T22:37:17.258489Z","title":"Cogagent: A visual language model for gui agents","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.258489Z"},"links":{"cited_paper":"/paper/2312.08914","citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:5f2d24d5ff7665c9f35a34fc4dd983146f2c5a9a2b5f61a8ba2e63de42f4e5ee","observation_id":"c556b550-d410-486b-afd0-d835d1f22636","resolution":{"observed_at":"2026-08-15T22:37:17.258489Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:18.363650Z","title":"Hudson and Christopher D","venue":null,"work_id":"11b6b6fb-75e0-444b-b1f8-0a00637e7bc9","year":2019},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.263432Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:720d792f6cd7644eb7ef15197edafaf30b5bbd2e84c07f5c232ecaec060f6196","observation_id":"079cc4b9-4d25-4ec3-a882-19c9707dce5f","resolution":{"observed_at":"2026-08-15T22:37:18.369113Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:18.346720Z","title":"Hénaff, Matthew M","venue":null,"work_id":"a7402ec6-d7d9-43ff-b8c9-b419679e373f","year":2022},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.268222Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:8e79a42f386bf00b076027b6d5ec110427597df79ccd4feb93d2da9ce825ec4f","observation_id":"c4697619-5555-48a5-bf31-fd9f83e3f969","resolution":{"observed_at":"2026-08-15T22:37:18.352433Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:18.331775Z","title":"Perceiver: General perception with iterative attention","venue":null,"work_id":"d3020799-1d3d-4424-9b48-668561750c22","year":2021},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.272925Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:011bb52c5dfc6a1dc0c89d2bda0eda81d6c34514055cc132f63a4e0b01475e47","observation_id":"b4bbc52a-509c-4bfe-8137-2d59fc846468","resolution":{"observed_at":"2026-08-15T22:37:18.336682Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:18.313303Z","title":null,"venue":null,"work_id":"fa1a5c69-975f-48aa-b65d-9485b2163573","year":2023},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.277455Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:4b91813f15516d84bbcce67f40496d7c4a311d48e8a57c3ca05ccb915eecbf08","observation_id":"3cf6af1a-dea7-4487-9cc0-4173795c0d7d","resolution":{"observed_at":"2026-08-15T22:37:18.320239Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:18.297293Z","title":null,"venue":null,"work_id":"3bb3706a-b2bf-40b4-b07f-8f518bb5b041","year":2024},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.282371Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:5afc645c0d4f32247eff2a74f1a06fc894ec27a13208aa1d9af25022a9145c66","observation_id":"591e322d-d301-4f49-86fa-f314e6369bdf","resolution":{"observed_at":"2026-08-15T22:37:18.302359Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:18.279955Z","title":"Dvqa: Understanding data visualizations via question answering","venue":null,"work_id":"52975367-5d8c-427e-8b6a-7fc1dd748eda","year":2018},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.287735Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:70cee29fd0300cac0795d42f6ee6ff7858be6c56a707891dbf62f295b959e257","observation_id":"d2f2cfe3-d13a-4d97-b3d3-996d296473a4","resolution":{"observed_at":"2026-08-15T22:37:18.285130Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1603.07396","last_updated":"2016-03-24T00:02:58Z","snapshot_observed_at":"2026-08-16T16:56:28.600506Z","submitted_at":"2016-03-24T00:02:58Z","title":"A Diagram Is Worth A Dozen Images","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1603.07396","snapshot_observed_at":"2026-08-15T22:37:17.292309Z","title":"A diagram is worth a dozen images","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.292309Z"},"links":{"cited_paper":"/paper/1603.07396","citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:6f9dabe3a297adc459768aa99652d5437f14b4897d9629b46a02770dede660a6","observation_id":"7843d8ba-0381-449d-8b5c-3cf1f1b5cd68","resolution":{"observed_at":"2026-08-15T22:37:17.292309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:18.259313Z","title":"Shamma, Michael S","venue":null,"work_id":"95664e33-5c33-4ebb-9dd5-d962f10ee37f","year":2017},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.297123Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:2c15edd4d85ee831e3d6c9e2af48ffcc6d8d299a3fc2ef7337703d7595fde885","observation_id":"8d53e7c6-25de-4c16-a365-66306cfef9fa","resolution":{"observed_at":"2026-08-15T22:37:18.265417Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:18.242682Z","title":"Rush, Douwe Kiela, Matthieu Cord, and Victor Sanh","venue":null,"work_id":"4006774e-d0f3-47b6-ba9d-fc23e639e5c1","year":2023},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.301607Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:dc99d04b580dd2c731b301c48bc59e1b4f95affefbe33f6fa1dc4f46b577f1a9","observation_id":"e1afaade-87ce-4d1e-bad3-bacf0af06bb2","resolution":{"observed_at":"2026-08-15T22:37:18.247940Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:18.225924Z","title":"Seed-bench: Benchmarking multimodal llms with generative comprehension, 2023","venue":null,"work_id":"c258a1c0-3a93-4a33-8ec7-1a951674506e","year":2023},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.306160Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:7ef577d7a281eef5dfad803532f20a3cc5b1c15d0756ff385027a046ec226da6","observation_id":"12794497-390f-4dd1-a131-303dcc767079","resolution":{"observed_at":"2026-08-15T22:37:18.231473Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.04219","last_updated":"2023-11-07T18:59:58Z","snapshot_observed_at":"2026-08-16T14:45:10.647424Z","submitted_at":"2023-11-07T18:59:58Z","title":"OtterHD: A High-Resolution Multi-modality Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.04219","snapshot_observed_at":"2026-08-15T22:37:17.310665Z","title":"Otterhd: A high- resolution multi-modality model","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.310665Z"},"links":{"cited_paper":"/paper/2311.04219","citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:29ee2a4bb34924a83af23f49a8e429b1e6f8e0389db335116a8a8507833cbfac","observation_id":"5ded6dda-4adb-48a2-b999-d05ffc3dd301","resolution":{"observed_at":"2026-08-15T22:37:17.310665Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:18.209895Z","title":null,"venue":null,"work_id":"ad59980e-3bab-43ef-a8d5-01331a5fd2a7","year":2023},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.315645Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:a05758e6a61c63466c948dff581ac3adf4f2b5c32b246a1728f2214d56540711","observation_id":"bb35ca61-05b3-4167-9e45-400fbba3f493","resolution":{"observed_at":"2026-08-15T22:37:18.214675Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03744","last_updated":"2024-05-15T19:22:44Z","snapshot_observed_at":"2026-08-15T02:35:59.111911Z","submitted_at":"2023-10-05T17:59:56Z","title":"Improved Baselines with Visual Instruction Tuning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03744","snapshot_observed_at":"2026-08-15T22:37:17.320912Z","title":"Improved baselines with visual instruction tuning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.320912Z"},"links":{"cited_paper":"/paper/2310.03744","citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:49c82aed507654991b59d23e7b7f9fbbcdf68c241c03d8b2dcb972472b7c8b1d","observation_id":"3e4290bf-ee0d-46d3-bfc7-554ede6c0307","resolution":{"observed_at":"2026-08-15T22:37:17.320912Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:18.193075Z","title":"Llava-next: Improved reasoning, ocr, and world knowledge, 2024","venue":null,"work_id":"e12f7c4a-a780-4435-8148-f2be70919e43","year":2024},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.325850Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:8e906f6f05a7ff0354d7c1721518cb35680b8a37d6d854f1f2fc09bdef6bf2c3","observation_id":"2443bf41-9797-4e95-826d-98b34ed56bf3","resolution":{"observed_at":"2026-08-15T22:37:18.198080Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:18.175022Z","title":"Visual instruction tuning","venue":null,"work_id":"274ac401-37a2-4003-a420-6fdd43e76f0b","year":2023},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.330537Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:046aeed6f1961542ccea66bad9a183a5d4a31ee901a2d1ec79eb0d4e446dea6f","observation_id":"1abdd7bc-d975-4a3d-b3ff-94a8c2699068","resolution":{"observed_at":"2026-08-15T22:37:18.180787Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.06281","last_updated":"2024-08-20T03:56:03Z","snapshot_observed_at":"2026-08-17T03:48:18.599994Z","submitted_at":"2023-07-12T16:23:09Z","title":"MMBench: Is Your Multi-modal Model an All-around Player?","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.06281","snapshot_observed_at":"2026-08-15T22:37:17.335514Z","title":"Mmbench: Is your multi-modal model an all-around player? arXiv:2307.06281, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.335514Z"},"links":{"cited_paper":"/paper/2307.06281","citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:1032d8fe0975ea460fcf1c79a4172733d4e48bb873efda0ff5ab85828f9284d0","observation_id":"accf8c68-4581-4dfe-a57e-3834d3bb63d9","resolution":{"observed_at":"2026-08-15T22:37:17.335514Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:18.157229Z","title":"UNIFIED-IO: A unified model for vision, language, and multi-modal tasks","venue":null,"work_id":"a5e9f16a-5cf2-45d4-9a15-700a56bc67a1","year":2023},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.340154Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:18993bd76d0f8c6f7e3bf04f65b634cb71d005c65f77f64cd302d9cd445af443","observation_id":"3de3b107-023e-49c7-9f93-5cc5b2c9b169","resolution":{"observed_at":"2026-08-15T22:37:18.162203Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:18.140583Z","title":"Mathvista: Evaluating mathematical reasoning of foundation models in visual contexts","venue":null,"work_id":"3fe9eadc-775a-4c0c-a20b-df8f5ac5cf3e","year":2024},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.344615Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:fc75e57b913e323baaec968792b1a9f2ee2a0922ad2bd6340440fa176d4cc4e2","observation_id":"a4311996-20f9-404d-ac32-e78646047c19","resolution":{"observed_at":"2026-08-15T22:37:18.145856Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:18.121755Z","title":"Learn to explain: Multimodal reasoning via thought chains for science question answering","venue":null,"work_id":"74228402-c94c-4930-a605-cb9baf18181c","year":2022},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.349090Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:d0a08b85c55f6cbff567d7d6121dcf995e13d0d114739c88b56d939dd86ad862","observation_id":"5d36567e-9a57-4797-a4db-8fc03cdcd8f2","resolution":{"observed_at":"2026-08-15T22:37:18.128235Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:18.105747Z","title":"A survivor in the era of large-scale pretraining: An empirical study of one-stage referring expression comprehension","venue":null,"work_id":"0ba1ecc6-9735-47e0-bef0-f2311f9b06df","year":2024},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.353541Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:c4b6c3cc6c4ab637c17db2e91e9236912ddd2ce5f893f534e54bf8041e5fcbb2","observation_id":"ffff0f23-5f26-451e-8e93-3658624d185b","resolution":{"observed_at":"2026-08-15T22:37:18.110682Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:18.090238Z","title":"Feast your eyes: Mixture-of-resolution adaptation for multimodal large language models, 2024","venue":null,"work_id":"f9a856f7-972b-4b02-a87a-8459ca86e2c3","year":2024},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.358115Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:a2b3ae743ca5549df6ab390f256d2616d11364ec341d8d39b183d16ce270b5ee","observation_id":"d394467a-a626-49c3-9447-95529818fc5a","resolution":{"observed_at":"2026-08-15T22:37:18.095311Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:17.362591Z","title":"Ok-vqa: A visual question answering benchmark requiring external knowledge","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.362591Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:4fef967306a12254623274b0b9afbda6b6deca83f1a141bbbbfcd64896a2b604","observation_id":"f5093c0c-ac0d-46d3-85b2-0fc5a29b8acd","resolution":{"observed_at":"2026-08-15T22:37:17.362591Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:18.062372Z","title":"ChartQA: A benchmark for question answering about charts with visual and logical reasoning","venue":null,"work_id":"f76be573-08c9-43f7-9e00-8241368e4539","year":2022},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.367414Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:6aadf2a4c4bb954a0f2fc304ec2c13b4e6f56a12b6aa494a137cd00ddb46cf64","observation_id":"643815db-9c99-4828-ad3b-482dac649944","resolution":{"observed_at":"2026-08-15T22:37:18.068053Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:18.046370Z","title":null,"venue":null,"work_id":"b3a631d8-ff19-4a72-bd86-f08e419ef5c2","year":2021},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.372229Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:d345a120d86a187f933bcb1e0ceef9b4f80cfd2a6f547350d8c665dca6e12d60","observation_id":"b4ce2632-ae02-4819-9afc-37b92cd0e315","resolution":{"observed_at":"2026-08-15T22:37:18.051687Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:17.376842Z","title":"Ocr-vqa: Visual question answering by reading text in images","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.376842Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:6d230130944a7046abd2ca338dd144fed2a958e75c3c9eb001d5c3ba8751a671","observation_id":"6a537e57-98fe-481b-93f2-3d14dccbf1ab","resolution":{"observed_at":"2026-08-15T22:37:17.376842Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:17.381421Z","title":"Gpt-4 technical report, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.381421Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:ade5b201e9208f1aaf804c005a9bf3d2204fde7b3a4b99fa941bd0d009892d3e","observation_id":"57f11aad-dc40-4278-9c8c-0a7153dafbc6","resolution":{"observed_at":"2026-08-15T22:37:17.381421Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:18.007681Z","title":"Learning transferable visual models from natural language supervision","venue":null,"work_id":"d24faf19-2d1e-49de-a1c7-b4606fdc753a","year":2021},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.385729Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:368b2cc43000754fa04ea02a8739b21579d7e5125c0d49f701f0a7088c488e4f","observation_id":"38811b5a-1a36-4c63-a18e-b5d8698edeed","resolution":{"observed_at":"2026-08-15T22:37:18.013848Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:17.991034Z","title":"Deepspeed: System optimizations enable training deep learning models with over 100 billion parameters","venue":null,"work_id":"1c50955d-9d64-4962-9592-f87d57cdc5d6","year":2020},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.390340Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:e917e8d01a4e63ea62f20963c70eeba023f17c5361fdf2d9f5f91dfcb21d3152","observation_id":"644f9a0f-3f3d-4334-af87-ca6ce011f14a","resolution":{"observed_at":"2026-08-15T22:37:17.997294Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:17.973739Z","title":"A-okvqa: A benchmark for visual question answering using world knowledge","venue":null,"work_id":"c0597648-11a5-4710-ab2a-e805b6ac11ec","year":2022},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.394934Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:d16b21d45f5b3cd5caaa23e85940642f54414be4f299a3eea702b009a2ee512e","observation_id":"fbc1b3c0-c457-481f-b7ca-e37b6a1f6c36","resolution":{"observed_at":"2026-08-15T22:37:17.979168Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:17.957616Z","title":"Unified model for image, video, audio and language tasks, 2023","venue":null,"work_id":"b313d086-07c8-4971-8f1e-5d995e3c9a63","year":2023},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.399666Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:fa59c931a79514dc01b6a41a490a8c6acb2963eeb4c3f9d2227b7c9f654c096c","observation_id":"dd55ed0c-9ba9-42fc-b859-2a79c31590df","resolution":{"observed_at":"2026-08-15T22:37:17.963174Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:17.941176Z","title":"Textcaps: a dataset for image captioningwith reading comprehension","venue":null,"work_id":"8898f404-5ee7-464a-bf1f-ff97fb4972ed","year":2020},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.403889Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:9b881472c049efc67438523a71c75985db38ab20d3b327d25de126aa163cb6c4","observation_id":"2528b9fb-0b48-43cc-b69d-6c13e0ac91c3","resolution":{"observed_at":"2026-08-15T22:37:17.946025Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:17.923910Z","title":"Towards vqa models that can read","venue":null,"work_id":"df9d28d3-177f-428e-ad25-1dd243d21cda","year":2019},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.408537Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:dbb482055a3d8ac6921c445e0176fa3832f531527e943717c5302971cdaee1aa","observation_id":"01a5cfba-bf17-49c6-9feb-8a83590156c9","resolution":{"observed_at":"2026-08-15T22:37:17.929242Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:17.907714Z","title":"Chameleon: Mixed-modal early-fusion foundation models","venue":null,"work_id":"16e8af32-031e-47bf-a15a-940c63af4f87","year":2024},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.412857Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:3705bcd77d80cff9fc8d6c3ab297f4cef110b5c3789e955b1f25e942337000ba","observation_id":"b1f1336e-84bb-485a-b837-13f6c6bf7ba7","resolution":{"observed_at":"2026-08-15T22:37:17.912465Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-15T22:37:17.417414Z","title":"Gemini: a family of highly capable multimodal models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.417414Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:5ca26b7760856d3e2e12ab4e82681140755aa6c4de9f4351fcaf257117c99003","observation_id":"81263de6-0bd7-480e-9e89-d89e75387bbf","resolution":{"observed_at":"2026-08-15T22:37:17.417414Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.06209","last_updated":"2024-04-25T07:12:39Z","snapshot_observed_at":"2026-08-16T14:28:07.420590Z","submitted_at":"2024-01-11T18:58:36Z","title":"Eyes Wide Shut? Exploring the Visual Shortcomings of Multimodal LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.06209","snapshot_observed_at":"2026-08-15T22:37:17.421900Z","title":"Eyes wide shut? exploring the visual shortcomings of multimodal llms","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.421900Z"},"links":{"cited_paper":"/paper/2401.06209","citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:40a99fbb1fe24fd98c45858938987ba7bdb84d93aed57d0309696fbef2fbf038","observation_id":"53f83287-8d9f-4f53-a903-57551cc69c45","resolution":{"observed_at":"2026-08-15T22:37:17.421900Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:17.891781Z","title":"Llama: Open and efficient foundation language models, 2023","venue":null,"work_id":"366b9238-571e-4d01-883a-b66a04b6f65a","year":2023},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.426731Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:67fdebefbccaaf2e5dee6a018b7251814c5e3f7ce213002c5486f90691011bb3","observation_id":"ec62cae5-d286-45dc-827e-d32018c112ae","resolution":{"observed_at":"2026-08-15T22:37:17.896807Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:17.874968Z","title":null,"venue":null,"work_id":"6dc71dbc-186d-4e53-bd04-b4f64642f844","year":2021},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.431671Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:7e7069e57a0beb035d1f7e3509093d35b07b0f330d207ad43bec30b4caa5833a","observation_id":"d5c8882a-c4d9-4b1a-a521-c035ec3b7c72","resolution":{"observed_at":"2026-08-15T22:37:17.880212Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:17.857587Z","title":"Ofa: Unifying architectures, tasks, and modalities through a simple sequence-to- sequence learning framework","venue":null,"work_id":"256b5f66-170e-4b80-8f89-b7108bdac593","year":2022},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.436174Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:d3c8a65ce4825f2504b98995ba385b21acd4a5387112607d3fc2db3ccb56d4c1","observation_id":"bd9ee375-7009-48ba-825e-079455b080cf","resolution":{"observed_at":"2026-08-15T22:37:17.862966Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.03079","last_updated":"2024-02-04T08:23:04Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-11-06T13:04:39Z","title":"CogVLM: Visual Expert for Pretrained Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.03079","snapshot_observed_at":"2026-08-15T22:37:17.440571Z","title":"Cogvlm: Visual expert for pretrained language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.440571Z"},"links":{"cited_paper":"/paper/2311.03079","citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:733a4b4fa9de5e460746a27a012c17c3e269591ca3bf7db48215251ca79dc478","observation_id":"f6c21bdc-d0a1-4ced-be6a-1080b311c758","resolution":{"observed_at":"2026-08-15T22:37:17.440571Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:17.840927Z","title":"Berg, and Tamara L","venue":null,"work_id":"5d94d8fe-35c7-4e9f-bb6a-c599df69965a","year":2016},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.445203Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:f2193866617e725e99230106da13e669a08d9a72b61e744236072631aa5d2990","observation_id":"e53a4edd-b4df-4dc6-b17a-9165e5a7d895","resolution":{"observed_at":"2026-08-15T22:37:17.846393Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:17.825862Z","title":"Mm-vet: Evaluating large multimodal models for integrated capabilities, 2023","venue":null,"work_id":"fa2f69e3-2f5a-43bb-80c2-da9e84f29794","year":2023},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.450074Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:57ab6b8bcb4714567a1ef88d628aeeba905a430d911acaf6493b5ea3e462cc78","observation_id":"13e047fc-b32f-47a1-81ff-3a562b728ff1","resolution":{"observed_at":"2026-08-15T22:37:17.830646Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:17.809374Z","title":"Mmmu: A massive multi-discipline multimodal understanding and reasoning benchmark for expert agi","venue":null,"work_id":"813841a9-9e40-43a4-b3b2-afc861d332cc","year":2024},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.455372Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:6647f91fc4c5a3e02a1969b664c11b427463a32aa54e94e05b7103f49cd70e6a","observation_id":"71e12d3c-17b0-4a0e-98be-a4dff0d7ba6e","resolution":{"observed_at":"2026-08-15T22:37:17.814495Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.16199","last_updated":"2024-09-18T23:54:36Z","snapshot_observed_at":"2026-08-13T10:46:50.834601Z","submitted_at":"2023-03-28T17:59:12Z","title":"LLaMA-Adapter: Efficient Fine-tuning of Language Models with Zero-init Attention","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.16199","snapshot_observed_at":"2026-08-15T22:37:17.460410Z","title":"Llama-adapter: Efficient fine-tuning of language models with zero-init attention","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.460410Z"},"links":{"cited_paper":"/paper/2303.16199","citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:4e1ee1be29d4af9100df94a8b2923d45d8800dcb8f4f284821ef615166bfcb86","observation_id":"dc659125-cc7c-43f0-a077-5dd32e9c7da3","resolution":{"observed_at":"2026-08-15T22:37:17.460410Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.10592","last_updated":"2023-10-02T16:38:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-04-20T18:25:35Z","title":"MiniGPT-4: Enhancing Vision-Language Understanding with Advanced Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.10592","snapshot_observed_at":"2026-08-15T22:37:17.465095Z","title":"Minigpt-4: Enhancing vision- language understanding with advanced large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.465095Z"},"links":{"cited_paper":"/paper/2304.10592","citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:1640b7b52eab4ed54ea30911a4cddb5c158bd83e8d6148a8a1562f13525c7954","observation_id":"664cb106-88ba-4595-a124-35a26f2fd489","resolution":{"observed_at":"2026-08-15T22:37:17.465095Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T22:37:17.790236Z","title":"Uni- perceiver: Pre-training unified architecture for generic perception for zero-shot and few-shot tasks","venue":null,"work_id":"f4cf5b45-ce4c-421c-8ba0-5d72d628871b","year":2022},"citing_paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-15T22:37:17.470659Z"},"links":{"citing_paper":"/paper/2505.06840"},"observation_digest":"sha256:bdad12c7d17f236ade56eed727154768984ec02032a7b78debce462952d23b07","observation_id":"625d38c0-d572-49d9-9c86-30596b0c8a6b","resolution":{"observed_at":"2026-08-15T22:37:17.797604Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-19T06:32:44.657259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2505.06840","last_updated":"2025-05-11T04:44:03Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-17T20:24:19.111791Z","submitted_at":"2025-05-11T04:44:03Z","title":"Visual Instruction Tuning with Chain of Region-of-Interest"},"reference_resolution":{"displayed":67,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":26,"verified_exact":0,"verified_fuzzy":41},"total_outbound_references":67},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-19T06:32:44.657259+00:00","source":"crossref"},{"observed_at":"2026-08-19T06:32:39.956319+00:00","source":"retraction_watch"}],"thesis":"As of 19 August 2026, this Paper Citation Record lists 67 of 67 outbound references and 0 inbound Pith citation observations for arXiv:2505.06840."}