{"as_of":"2026-08-04T23:29:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:614f89f287d221df7da9de76b41d9737e8e8a5bf17df78c421acd82bd9a861a9","coverage":[{"denominator":34,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":34,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-22T13:29:47.529564Z","state":"measured"},{"denominator":68,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":68,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-04T06:34:03.388597+00:00","state":"measured"},{"denominator":34,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":34,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T05:31:41.728219Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-11T03:17:51.507547Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"cited_work":{"arxiv_id":"2505.15879","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.15879","snapshot_observed_at":"2026-07-11T03:17:51.507547Z","title":"GRIT: Teaching MLLMs to Think with Images","venue":"cs.CV","work_id":"0a45eb3d-cd3b-4535-9626-ac689cfff7e1","year":2025},"citing_paper":{"arxiv_id":"2509.02547","last_updated":"2026-04-17T18:09:08Z","snapshot_observed_at":"2026-08-03T09:07:42.489237Z","submitted_at":"2025-09-02T17:46:26Z","title":"The Landscape of Agentic Reinforcement Learning for LLMs: A Survey","version":5},"reference_index":246,"source":"pdf_text","source_observed_at":"2026-05-18T19:19:36.427337Z"},"links":{"cited_paper":"/paper/2505.15879","citing_paper":"/paper/2509.02547"},"observation_digest":"sha256:19096e60d3f8f642bf5a7239a39ea71724a44081d3ec4aace37f730776147571","observation_id":"05b20e05-7c95-4719-84c9-a843862c2d35","resolution":{"observed_at":"2026-05-18T19:21:48.385988Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"cited_work":{"arxiv_id":"2505.15879","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.15879","snapshot_observed_at":"2026-07-11T03:17:51.507547Z","title":"GRIT: Teaching MLLMs to Think with Images","venue":"cs.CV","work_id":"0a45eb3d-cd3b-4535-9626-ac689cfff7e1","year":2025},"citing_paper":{"arxiv_id":"2509.08827","last_updated":"2025-10-09T17:08:52Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-09-10T17:59:43Z","title":"A Survey of Reinforcement Learning for Large Reasoning Models","version":3},"reference_index":133,"source":"arxiv_source","source_observed_at":"2026-05-18T00:02:24.352947Z"},"links":{"cited_paper":"/paper/2505.15879","citing_paper":"/paper/2509.08827"},"observation_digest":"sha256:29c140110d1708e64618da5040e63d9c224c3b87ef556a52bd10855766f2cad6","observation_id":"8f2e4bf3-41ee-4dfb-9f20-333b58d1e1ed","resolution":{"observed_at":"2026-05-18T00:02:25.351953Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"cited_work":{"arxiv_id":"2505.15879","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.15879","snapshot_observed_at":"2026-07-11T03:17:51.507547Z","title":"GRIT: Teaching MLLMs to Think with Images","venue":"cs.CV","work_id":"0a45eb3d-cd3b-4535-9626-ac689cfff7e1","year":2025},"citing_paper":{"arxiv_id":"2509.20912","last_updated":"2026-05-21T02:54:31Z","snapshot_observed_at":"2026-07-06T22:30:46.129271Z","submitted_at":"2025-09-25T08:58:10Z","title":"DeFacto: Counterfactual Thinking with Images for Enforcing Evidence-Grounded and Faithful Reasoning","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-21T22:35:36.136639Z"},"links":{"cited_paper":"/paper/2505.15879","citing_paper":"/paper/2509.20912"},"observation_digest":"sha256:06562c10eccc9916dfc88db68c06cffab1113723671f9f163e263cdea24b923c","observation_id":"2165189f-3d1f-4442-9437-75c4bef659ef","resolution":{"observed_at":"2026-05-21T22:35:42.653389Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"cited_work":{"arxiv_id":"2505.15879","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.15879","snapshot_observed_at":"2026-07-11T03:17:51.507547Z","title":"GRIT: Teaching MLLMs to Think with Images","venue":"cs.CV","work_id":"0a45eb3d-cd3b-4535-9626-ac689cfff7e1","year":2025},"citing_paper":{"arxiv_id":"2509.20912","last_updated":"2026-05-21T02:54:31Z","snapshot_observed_at":"2026-07-06T22:30:46.129271Z","submitted_at":"2025-09-25T08:58:10Z","title":"DeFacto: Counterfactual Thinking with Images for Enforcing Evidence-Grounded and Faithful Reasoning","version":4},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-22T13:36:06.229352Z"},"links":{"cited_paper":"/paper/2505.15879","citing_paper":"/paper/2509.20912"},"observation_digest":"sha256:e0f9e94259eda5d3adcf3a736349d393d3063cad8b6e4a9bbb13f2e52738386a","observation_id":"9decbf5c-c00d-4215-9bd4-540abe876579","resolution":{"observed_at":"2026-05-22T13:36:35.906344Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"cited_work":{"arxiv_id":"2505.15879","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.15879","snapshot_observed_at":"2026-07-11T03:17:51.507547Z","title":"GRIT: Teaching MLLMs to Think with Images","venue":"cs.CV","work_id":"0a45eb3d-cd3b-4535-9626-ac689cfff7e1","year":2025},"citing_paper":{"arxiv_id":"2509.22746","last_updated":"2026-05-14T08:41:49Z","snapshot_observed_at":"2026-07-31T03:14:47.894025Z","submitted_at":"2025-09-26T04:33:53Z","title":"Mixture-of-Visual-Thoughts: Exploring Context-Adaptive Reasoning Mode Selection for General Visual Reasoning","version":2},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-05-18T13:33:12.508639Z"},"links":{"cited_paper":"/paper/2505.15879","citing_paper":"/paper/2509.22746"},"observation_digest":"sha256:e6e406997454dff20f46f4a5296ac204b121081e833799d7f9ab4f9204c40d81","observation_id":"9c9dc810-3210-4559-aafd-a0dafc2aa787","resolution":{"observed_at":"2026-05-18T13:36:25.109996Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"cited_work":{"arxiv_id":"2505.15879","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.15879","snapshot_observed_at":"2026-07-11T03:17:51.507547Z","title":"GRIT: Teaching MLLMs to Think with Images","venue":"cs.CV","work_id":"0a45eb3d-cd3b-4535-9626-ac689cfff7e1","year":2025},"citing_paper":{"arxiv_id":"2510.04225","last_updated":"2026-04-21T18:16:55Z","snapshot_observed_at":"2026-07-06T22:31:44.964774Z","submitted_at":"2025-10-05T14:29:01Z","title":"Locate-Then-Examine: Grounded Region Reasoning Improves Detection of AI-Generated Images","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-18T10:13:08.112072Z"},"links":{"cited_paper":"/paper/2505.15879","citing_paper":"/paper/2510.04225"},"observation_digest":"sha256:4e257189aee9f16f9cb839fc35957dc1a77ec96392dde39673fad22b30753624","observation_id":"8b981fc6-76b5-4a4c-a00b-8a8519489161","resolution":{"observed_at":"2026-05-18T10:16:14.380167Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"cited_work":{"arxiv_id":"2505.15879","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.15879","snapshot_observed_at":"2026-07-11T03:17:51.507547Z","title":"GRIT: Teaching MLLMs to Think with Images","venue":"cs.CV","work_id":"0a45eb3d-cd3b-4535-9626-ac689cfff7e1","year":2025},"citing_paper":{"arxiv_id":"2511.05271","last_updated":"2026-03-11T08:46:41Z","snapshot_observed_at":"2026-07-06T22:35:13.699594Z","submitted_at":"2025-11-07T14:31:20Z","title":"DeepEyesV2: Toward Agentic Multimodal Model","version":4},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-16T05:32:29.266583Z"},"links":{"cited_paper":"/paper/2505.15879","citing_paper":"/paper/2511.05271"},"observation_digest":"sha256:afea17e56103c9701b4aaaa4714d45b91cfe8caaacfd2a56c87b931bb8447976","observation_id":"8243c96f-1865-430b-97fa-0f15d800c5b2","resolution":{"observed_at":"2026-05-16T05:32:29.551334Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"cited_work":{"arxiv_id":"2505.15879","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.15879","snapshot_observed_at":"2026-07-11T03:17:51.507547Z","title":"GRIT: Teaching MLLMs to Think with Images","venue":"cs.CV","work_id":"0a45eb3d-cd3b-4535-9626-ac689cfff7e1","year":2025},"citing_paper":{"arxiv_id":"2512.03438","last_updated":"2026-07-30T22:27:17Z","snapshot_observed_at":"2026-08-04T23:10:05.405788Z","submitted_at":"2025-12-03T04:42:47Z","title":"Multimodal Reinforcement Learning with Adaptive Verifier for AI Agents","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-17T03:09:31.161760Z"},"links":{"cited_paper":"/paper/2505.15879","citing_paper":"/paper/2512.03438"},"observation_digest":"sha256:10a1d0684e781b98cddfb68b063e3272ccb6e0e0cd4a74105ad0f30ad9c27855","observation_id":"2c11580c-89a5-4c9b-a428-0d40fa656a99","resolution":{"observed_at":"2026-05-17T03:11:30.074062Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.15879","snapshot_observed_at":"2026-08-03T18:51:42.710473Z","title":"Grit: Teaching mllms to think with images.arXiv preprint arXiv:2505.15879, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2512.03438","last_updated":"2026-07-30T22:27:17Z","snapshot_observed_at":"2026-08-04T23:10:05.405788Z","submitted_at":"2025-12-03T04:42:47Z","title":"Multimodal Reinforcement Learning with Adaptive Verifier for AI Agents","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-03T18:51:42.710473Z"},"links":{"cited_paper":"/paper/2505.15879","citing_paper":"/paper/2512.03438"},"observation_digest":"sha256:3be22e3ee40fe1fd7c3542af1f3f4afce934c0561021fd08305d29aff893f333","observation_id":"cad19910-7fe8-4123-a681-b87d362055f9","resolution":{"observed_at":"2026-08-03T18:51:42.710473Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"cited_work":{"arxiv_id":"2505.15879","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.15879","snapshot_observed_at":"2026-07-11T03:17:51.507547Z","title":"GRIT: Teaching MLLMs to Think with Images","venue":"cs.CV","work_id":"0a45eb3d-cd3b-4535-9626-ac689cfff7e1","year":2025},"citing_paper":{"arxiv_id":"2512.12623","last_updated":"2026-04-09T15:39:45Z","snapshot_observed_at":"2026-07-06T22:38:59.725645Z","submitted_at":"2025-12-14T10:07:45Z","title":"Reasoning Within the Mind: Dynamic Multimodal Interleaving in Latent Space","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-16T23:02:28.588225Z"},"links":{"cited_paper":"/paper/2505.15879","citing_paper":"/paper/2512.12623"},"observation_digest":"sha256:fa2619fd136c796ce122a411e6e29cc0eb661c03a97737e4838d1037f2bb8653","observation_id":"c7a5c95e-ca17-4949-bfcf-e30dd4b45186","resolution":{"observed_at":"2026-05-16T23:03:38.976265Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.15879","snapshot_observed_at":"2026-08-03T05:23:31.745845Z","title":"J., Guan, X., and Wang, X","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.02465","last_updated":"2026-06-10T12:52:04Z","snapshot_observed_at":"2026-08-03T05:23:28.259581Z","submitted_at":"2026-02-02T18:49:06Z","title":"MentisOculi: Revealing the Limits of Reasoning with Mental Imagery","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-03T05:23:31.745845Z"},"links":{"cited_paper":"/paper/2505.15879","citing_paper":"/paper/2602.02465"},"observation_digest":"sha256:46ddacaaffa974d042f9ed73b783efc2dbc93e87937e76c6452f1ece4f028366","observation_id":"09682d13-051c-4bff-9f05-6eeffeabaf4a","resolution":{"observed_at":"2026-08-03T05:23:31.745845Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"cited_work":{"arxiv_id":"2505.15879","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.15879","snapshot_observed_at":"2026-07-11T03:17:51.507547Z","title":"GRIT: Teaching MLLMs to Think with Images","venue":"cs.CV","work_id":"0a45eb3d-cd3b-4535-9626-ac689cfff7e1","year":2025},"citing_paper":{"arxiv_id":"2604.02794","last_updated":"2026-04-03T07:02:13Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-03T07:02:13Z","title":"CharTool: Tool-Integrated Visual Reasoning for Chart Understanding","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-05-13T20:07:23.153064Z"},"links":{"cited_paper":"/paper/2505.15879","citing_paper":"/paper/2604.02794"},"observation_digest":"sha256:31c51ff6c42d7b6e179f6ac355b58bc10b6f3dc4257e80c1a51bbdd1016080c5","observation_id":"e3983524-ce36-4706-9ef6-481081bf2eab","resolution":{"observed_at":"2026-05-13T20:08:12.734245Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"cited_work":{"arxiv_id":"2505.15879","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.15879","snapshot_observed_at":"2026-07-11T03:17:51.507547Z","title":"GRIT: Teaching MLLMs to Think with Images","venue":"cs.CV","work_id":"0a45eb3d-cd3b-4535-9626-ac689cfff7e1","year":2025},"citing_paper":{"arxiv_id":"2604.04733","last_updated":"2026-04-24T20:17:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-06T15:00:54Z","title":"Discovering Failure Modes in Vision-Language Models using RL","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-10T19:21:25.761342Z"},"links":{"cited_paper":"/paper/2505.15879","citing_paper":"/paper/2604.04733"},"observation_digest":"sha256:95337de279b379a130ec0518204d18179b4dd812f175d1f05e3b8caf4a0d6fc9","observation_id":"2153c1ed-3c36-4215-9068-cd12049a8599","resolution":{"observed_at":"2026-05-12T02:44:46.816923Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"cited_work":{"arxiv_id":"2505.15879","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.15879","snapshot_observed_at":"2026-07-11T03:17:51.507547Z","title":"GRIT: Teaching MLLMs to Think with Images","venue":"cs.CV","work_id":"0a45eb3d-cd3b-4535-9626-ac689cfff7e1","year":2025},"citing_paper":{"arxiv_id":"2604.06079","last_updated":"2026-04-07T16:58:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-07T16:58:14Z","title":"Scientific Graphics Program Synthesis via Dual Self-Consistency Reinforcement Learning","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-10T19:45:40.915428Z"},"links":{"cited_paper":"/paper/2505.15879","citing_paper":"/paper/2604.06079"},"observation_digest":"sha256:738094ab21b39c9f3de86867ff52afdf5911669633dc1f767680e0bafdaa3aaa","observation_id":"9d39e6dc-9789-4e6a-b8d1-72d728f27bfd","resolution":{"observed_at":"2026-05-12T02:44:46.816923Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"cited_work":{"arxiv_id":"2505.15879","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.15879","snapshot_observed_at":"2026-07-11T03:17:51.507547Z","title":"GRIT: Teaching MLLMs to Think with Images","venue":"cs.CV","work_id":"0a45eb3d-cd3b-4535-9626-ac689cfff7e1","year":2025},"citing_paper":{"arxiv_id":"2604.11025","last_updated":"2026-08-02T20:03:04Z","snapshot_observed_at":"2026-08-04T23:28:59.036170Z","submitted_at":"2026-04-13T05:49:04Z","title":"Test-time Scaling over Perception: Resolving the Grounding Paradox in Thinking with Images","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-10T16:38:11.785469Z"},"links":{"cited_paper":"/paper/2505.15879","citing_paper":"/paper/2604.11025"},"observation_digest":"sha256:dbd2852f2d21a0641f49700152e9c52ad321f6597e677285ec91695e26a017b1","observation_id":"d6277388-5c56-4714-886c-2f73cd00da37","resolution":{"observed_at":"2026-05-12T02:44:46.816923Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.15879","snapshot_observed_at":"2026-08-04T05:31:41.728219Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2604.11025","last_updated":"2026-08-02T20:03:04Z","snapshot_observed_at":"2026-08-04T23:28:59.036170Z","submitted_at":"2026-04-13T05:49:04Z","title":"Test-time Scaling over Perception: Resolving the Grounding Paradox in Thinking with Images","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-04T05:31:41.728219Z"},"links":{"cited_paper":"/paper/2505.15879","citing_paper":"/paper/2604.11025"},"observation_digest":"sha256:b8d9cc505a98675ae37f222a4b1dd208f50c35e5c017d4b333f95da4d3b75c34","observation_id":"9871af4a-fca1-4444-bf8f-d4faaf0f86f8","resolution":{"observed_at":"2026-08-04T05:31:41.728219Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"cited_work":{"arxiv_id":"2505.15879","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.15879","snapshot_observed_at":"2026-07-11T03:17:51.507547Z","title":"GRIT: Teaching MLLMs to Think with Images","venue":"cs.CV","work_id":"0a45eb3d-cd3b-4535-9626-ac689cfff7e1","year":2025},"citing_paper":{"arxiv_id":"2604.12896","last_updated":"2026-04-14T15:45:22Z","snapshot_observed_at":"2026-08-04T17:57:48.307510Z","submitted_at":"2026-04-14T15:45:22Z","title":"Don't Show Pixels, Show Cues: Unlocking Visual Tool Reasoning in Language Models via Perception Programs","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-10T15:29:25.650175Z"},"links":{"cited_paper":"/paper/2505.15879","citing_paper":"/paper/2604.12896"},"observation_digest":"sha256:f992532ec69fb11345a2cc1e58f2e2d676811474570b44fa8ac08e223629d5e8","observation_id":"e2acb4af-17dc-4eab-ab7c-239ce84866ee","resolution":{"observed_at":"2026-05-12T02:44:46.816923Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"cited_work":{"arxiv_id":"2505.15879","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.15879","snapshot_observed_at":"2026-07-11T03:17:51.507547Z","title":"GRIT: Teaching MLLMs to Think with Images","venue":"cs.CV","work_id":"0a45eb3d-cd3b-4535-9626-ac689cfff7e1","year":2025},"citing_paper":{"arxiv_id":"2604.16054","last_updated":"2026-04-17T13:29:46Z","snapshot_observed_at":"2026-08-02T08:15:43.582880Z","submitted_at":"2026-04-17T13:29:46Z","title":"Mind's Eye: A Benchmark of Visual Abstraction, Transformation and Composition for Multimodal LLMs","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T09:18:33.234354Z"},"links":{"cited_paper":"/paper/2505.15879","citing_paper":"/paper/2604.16054"},"observation_digest":"sha256:2cc6eb7ca84ceca930707cddd0129653f78633fdb03a2c178e868a5b3b85d4a1","observation_id":"eec7037c-3b7d-4415-9e17-781290b83c91","resolution":{"observed_at":"2026-05-12T02:44:46.816923Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"cited_work":{"arxiv_id":"2505.15879","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.15879","snapshot_observed_at":"2026-07-11T03:17:51.507547Z","title":"GRIT: Teaching MLLMs to Think with Images","venue":"cs.CV","work_id":"0a45eb3d-cd3b-4535-9626-ac689cfff7e1","year":2025},"citing_paper":{"arxiv_id":"2605.20165","last_updated":"2026-05-19T17:50:25Z","snapshot_observed_at":"2026-07-06T23:30:49.211483Z","submitted_at":"2026-05-19T17:50:25Z","title":"CaMo: Camera Motion Grounded Evaluation and Training for Vision-Language Models","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-05-20T05:27:30.938311Z"},"links":{"cited_paper":"/paper/2505.15879","citing_paper":"/paper/2605.20165"},"observation_digest":"sha256:5633e1d3910033c52bcc6c3e8ed4d8b8b48e0c28df5d5053456adb6a6b541c9e","observation_id":"282fc9d3-ed0e-4203-9244-9bfbfc1e3a45","resolution":{"observed_at":"2026-05-20T05:28:04.625613Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"cited_work":{"arxiv_id":"2505.15879","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.15879","snapshot_observed_at":"2026-07-11T03:17:51.507547Z","title":"GRIT: Teaching MLLMs to Think with Images","venue":"cs.CV","work_id":"0a45eb3d-cd3b-4535-9626-ac689cfff7e1","year":2025},"citing_paper":{"arxiv_id":"2605.21652","last_updated":"2026-05-23T13:26:54Z","snapshot_observed_at":"2026-07-06T23:32:05.693503Z","submitted_at":"2026-05-20T19:06:34Z","title":"Look-Closer-Then-Diagnose: Confidence-Aware Ultrasound VQA via Active Zooming","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-22T09:13:45.830192Z"},"links":{"cited_paper":"/paper/2505.15879","citing_paper":"/paper/2605.21652"},"observation_digest":"sha256:3f843a15e1b5b9f0e4023fcff7518c243d029b2b1c850d6562a4b5872e962e2f","observation_id":"b110cb73-25d9-4b4d-899a-6d4b6e567fbb","resolution":{"observed_at":"2026-05-22T09:14:45.465091Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"cited_work":{"arxiv_id":"2505.15879","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.15879","snapshot_observed_at":"2026-07-11T03:17:51.507547Z","title":"GRIT: Teaching MLLMs to Think with Images","venue":"cs.CV","work_id":"0a45eb3d-cd3b-4535-9626-ac689cfff7e1","year":2025},"citing_paper":{"arxiv_id":"2605.21652","last_updated":"2026-05-23T13:26:54Z","snapshot_observed_at":"2026-07-06T23:32:05.693503Z","submitted_at":"2026-05-20T19:06:34Z","title":"Look-Closer-Then-Diagnose: Confidence-Aware Ultrasound VQA via Active Zooming","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-30T16:48:55.898188Z"},"links":{"cited_paper":"/paper/2505.15879","citing_paper":"/paper/2605.21652"},"observation_digest":"sha256:67de8a3c8b8daa90dae91e6c3d1ecd139c1ad2c53f778162c99ff6993bc117ff","observation_id":"01417436-c34e-4ec4-8566-57ecca9d9828","resolution":{"observed_at":"2026-06-30T16:54:58.761951Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"cited_work":{"arxiv_id":"2505.15879","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.15879","snapshot_observed_at":"2026-07-11T03:17:51.507547Z","title":"GRIT: Teaching MLLMs to Think with Images","venue":"cs.CV","work_id":"0a45eb3d-cd3b-4535-9626-ac689cfff7e1","year":2025},"citing_paper":{"arxiv_id":"2605.27960","last_updated":"2026-05-27T04:54:07Z","snapshot_observed_at":"2026-08-03T00:24:19.442122Z","submitted_at":"2026-05-27T04:54:07Z","title":"Mags-RL: Wearing Multimodal LLMs a Magnifying Glass via Agentic Reinforcement Learning For Complex Scene Reasoning","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-06-29T13:38:01.819121Z"},"links":{"cited_paper":"/paper/2505.15879","citing_paper":"/paper/2605.27960"},"observation_digest":"sha256:1b37df937989ab6ce3f16f34ffb89b2cfd7aa85c82ff9752b83082206797dd20","observation_id":"2f17d4ff-7439-4adc-b822-908ea9db3231","resolution":{"observed_at":"2026-06-29T13:43:29.025496Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"cited_work":{"arxiv_id":"2505.15879","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.15879","snapshot_observed_at":"2026-07-11T03:17:51.507547Z","title":"GRIT: Teaching MLLMs to Think with Images","venue":"cs.CV","work_id":"0a45eb3d-cd3b-4535-9626-ac689cfff7e1","year":2025},"citing_paper":{"arxiv_id":"2605.28741","last_updated":"2026-05-27T17:01:39Z","snapshot_observed_at":"2026-07-06T23:38:19.096349Z","submitted_at":"2026-05-27T17:01:39Z","title":"Self-Prophetic Decoding to Unlock Visual Search in LVLMs","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-29T12:53:40.783281Z"},"links":{"cited_paper":"/paper/2505.15879","citing_paper":"/paper/2605.28741"},"observation_digest":"sha256:8138512ae740d0ef5a5e512cb4a5cf1fffe24db655f312931677ac62ca0b8b5d","observation_id":"6e4cb97b-624b-43f8-ac6b-4f2b5583ead8","resolution":{"observed_at":"2026-06-29T13:03:26.930383Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"cited_work":{"arxiv_id":"2505.15879","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.15879","snapshot_observed_at":"2026-07-11T03:17:51.507547Z","title":"GRIT: Teaching MLLMs to Think with Images","venue":"cs.CV","work_id":"0a45eb3d-cd3b-4535-9626-ac689cfff7e1","year":2025},"citing_paper":{"arxiv_id":"2606.01249","last_updated":"2026-06-17T04:44:10Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-05-31T14:04:51Z","title":"Trust Region On-Policy Distillation","version":3},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-06-28T17:38:50.313305Z"},"links":{"cited_paper":"/paper/2505.15879","citing_paper":"/paper/2606.01249"},"observation_digest":"sha256:c7ee746c3fc3bc49a96311331317ddfd99678cbbfda12f2b5408c97e600db251","observation_id":"d771a75e-8bff-4560-9352-0dabae6d2137","resolution":{"observed_at":"2026-07-01T20:56:13.504175Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"cited_work":{"arxiv_id":"2505.15879","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.15879","snapshot_observed_at":"2026-07-11T03:17:51.507547Z","title":"GRIT: Teaching MLLMs to Think with Images","venue":"cs.CV","work_id":"0a45eb3d-cd3b-4535-9626-ac689cfff7e1","year":2025},"citing_paper":{"arxiv_id":"2606.11740","last_updated":"2026-06-10T07:16:27Z","snapshot_observed_at":"2026-08-02T02:17:16.809620Z","submitted_at":"2026-06-10T07:16:27Z","title":"UniReason-Med: A Shared Grounded Reasoning Interface for 2D-to-3D Transfer in Medical VQA","version":1},"reference_index":129,"source":"arxiv_source","source_observed_at":"2026-06-27T10:21:12.782864Z"},"links":{"cited_paper":"/paper/2505.15879","citing_paper":"/paper/2606.11740"},"observation_digest":"sha256:c25218c8772020f56f61466d3d1c4ad5f018ac81fa4658ce0575d08c44a6a7b0","observation_id":"c854a493-ce9f-4840-8c0a-4fd43b61c188","resolution":{"observed_at":"2026-07-03T09:47:59.480737Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"cited_work":{"arxiv_id":"2505.15879","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.15879","snapshot_observed_at":"2026-07-11T03:17:51.507547Z","title":"GRIT: Teaching MLLMs to Think with Images","venue":"cs.CV","work_id":"0a45eb3d-cd3b-4535-9626-ac689cfff7e1","year":2025},"citing_paper":{"arxiv_id":"2606.24118","last_updated":"2026-06-23T04:09:05Z","snapshot_observed_at":"2026-08-03T04:47:08.013790Z","submitted_at":"2026-06-23T04:09:05Z","title":"An LMM for Precisely Grounding Elements in Documents","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-26T01:51:50.043216Z"},"links":{"cited_paper":"/paper/2505.15879","citing_paper":"/paper/2606.24118"},"observation_digest":"sha256:9d15bb7bea042c055e9eed5b4ba0097f81ed7e9d5139fb71498a71e085482982","observation_id":"a11129eb-b7b5-43b5-8421-b2d0e1606be4","resolution":{"observed_at":"2026-07-04T15:09:54.669584Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"cited_work":{"arxiv_id":"2505.15879","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.15879","snapshot_observed_at":"2026-07-11T03:17:51.507547Z","title":"GRIT: Teaching MLLMs to Think with Images","venue":"cs.CV","work_id":"0a45eb3d-cd3b-4535-9626-ac689cfff7e1","year":2025},"citing_paper":{"arxiv_id":"2606.25319","last_updated":"2026-06-24T02:32:38Z","snapshot_observed_at":"2026-08-04T23:24:39.482468Z","submitted_at":"2026-06-24T02:32:38Z","title":"V-Zero: Answer-Label-Free On-Policy Distillation with Contrastive Evidence Gating for Fine-Grained Visual Reasoning","version":1},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-06-25T21:23:10.051805Z"},"links":{"cited_paper":"/paper/2505.15879","citing_paper":"/paper/2606.25319"},"observation_digest":"sha256:d8d800f653a7ab9e9961d0524a4d1de81ebb317dc8dc3c822eb05181bf7f99a2","observation_id":"dabee48d-3830-4dc2-8591-6ae2e5417b62","resolution":{"observed_at":"2026-07-04T19:30:06.787705Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"cited_work":{"arxiv_id":"2505.15879","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.15879","snapshot_observed_at":"2026-07-11T03:17:51.507547Z","title":"GRIT: Teaching MLLMs to Think with Images","venue":"cs.CV","work_id":"0a45eb3d-cd3b-4535-9626-ac689cfff7e1","year":2025},"citing_paper":{"arxiv_id":"2606.26196","last_updated":"2026-06-24T15:20:32Z","snapshot_observed_at":"2026-07-07T00:00:31.981360Z","submitted_at":"2026-06-24T15:20:32Z","title":"From Structure to Synergy: A Survey of Vision-Language Perception Paradigm Evolution in Multimodal Large Language Models","version":1},"reference_index":177,"source":"pdf_text","source_observed_at":"2026-06-26T01:50:54.242508Z"},"links":{"cited_paper":"/paper/2505.15879","citing_paper":"/paper/2606.26196"},"observation_digest":"sha256:1f5c3800985705d3108ca88aa1d389b6293d5bf0bc179707e12180f15f49b06a","observation_id":"2ebcddd0-050f-4521-81ad-2fda7b48cd47","resolution":{"observed_at":"2026-07-04T15:09:54.988860Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"cited_work":{"arxiv_id":"2505.15879","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.15879","snapshot_observed_at":"2026-07-11T03:17:51.507547Z","title":"GRIT: Teaching MLLMs to Think with Images","venue":"cs.CV","work_id":"0a45eb3d-cd3b-4535-9626-ac689cfff7e1","year":2025},"citing_paper":{"arxiv_id":"2606.30168","last_updated":"2026-06-29T11:44:09Z","snapshot_observed_at":"2026-07-07T00:04:05.275293Z","submitted_at":"2026-06-29T11:44:09Z","title":"Latent Noise Mask for Reducing Visual Redundancy in Multimodal Large Language Models","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-30T06:35:16.868238Z"},"links":{"cited_paper":"/paper/2505.15879","citing_paper":"/paper/2606.30168"},"observation_digest":"sha256:ac2d423f67db8834db5ac22210330605e1db72f6d64dd0a435100eb25e0f5b06","observation_id":"ed133d53-aeb3-45e2-9415-f99e1dfba12a","resolution":{"observed_at":"2026-06-30T06:54:21.214939Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"cited_work":{"arxiv_id":"2505.15879","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.15879","snapshot_observed_at":"2026-07-11T03:17:51.507547Z","title":"GRIT: Teaching MLLMs to Think with Images","venue":"cs.CV","work_id":"0a45eb3d-cd3b-4535-9626-ac689cfff7e1","year":2025},"citing_paper":{"arxiv_id":"2607.00881","last_updated":"2026-07-01T12:45:12Z","snapshot_observed_at":"2026-08-03T01:24:10.985411Z","submitted_at":"2026-07-01T12:45:12Z","title":"OmniView-Space: Reinforcing Spatial Reasoning via Multi-Perspective Spatial Mapping","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-07-02T14:16:39.649823Z"},"links":{"cited_paper":"/paper/2505.15879","citing_paper":"/paper/2607.00881"},"observation_digest":"sha256:fc9eec5f69a5624c50ccd3d5dc503a18f1b5033644fe966a35ef0ef7a7e63cf3","observation_id":"fb371fa9-c51b-432c-b8b9-6e9e36278253","resolution":{"observed_at":"2026-07-02T14:17:02.390573Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"cited_work":{"arxiv_id":"2505.15879","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.15879","snapshot_observed_at":"2026-07-11T03:17:51.507547Z","title":"GRIT: Teaching MLLMs to Think with Images","venue":"cs.CV","work_id":"0a45eb3d-cd3b-4535-9626-ac689cfff7e1","year":2025},"citing_paper":{"arxiv_id":"2607.01191","last_updated":"2026-07-01T17:24:26Z","snapshot_observed_at":"2026-08-02T15:51:49.700376Z","submitted_at":"2026-07-01T17:24:26Z","title":"Perceive-to-Reason: Decoupling Perception and Reasoning for Fine-Grained Visual Reasoning","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-07-02T13:24:17.538850Z"},"links":{"cited_paper":"/paper/2505.15879","citing_paper":"/paper/2607.01191"},"observation_digest":"sha256:57a41e2c0f7f50fa7eef17ab075b1e92bc58cb2a53ad5dff47b51c231a9ee551","observation_id":"96380a47-4cd1-4d20-b38c-26b01ec491ec","resolution":{"observed_at":"2026-07-02T13:26:58.445181Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.15879","snapshot_observed_at":"2026-07-11T19:17:19.634242Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.04423","last_updated":"2026-07-07T07:27:22Z","snapshot_observed_at":"2026-08-04T06:57:07.480777Z","submitted_at":"2026-07-05T17:33:59Z","title":"Transferability Between Understanding and Generation in Unified Multimodal Models","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-07-11T19:17:19.634242Z"},"links":{"cited_paper":"/paper/2505.15879","citing_paper":"/paper/2607.04423"},"observation_digest":"sha256:f152558b76a8abac889fd98dfe4b8daffa1a4709a71cdc63b0e17edc864b6299","observation_id":"017bab76-2cb8-4f60-864c-553c7b794daf","resolution":{"observed_at":"2026-07-11T19:17:19.634242Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"cited_work":{"arxiv_id":"2505.15879","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.15879","snapshot_observed_at":"2026-07-11T03:17:51.507547Z","title":"GRIT: Teaching MLLMs to Think with Images","venue":"cs.CV","work_id":"0a45eb3d-cd3b-4535-9626-ac689cfff7e1","year":2025},"citing_paper":{"arxiv_id":"2607.05716","last_updated":"2026-07-13T06:34:36Z","snapshot_observed_at":"2026-08-02T09:35:49.380987Z","submitted_at":"2026-07-07T01:00:51Z","title":"Scene Graph Thinking: Reinforcing Structured Visual Reasoning for Multimodal Large Language Models","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-07-11T03:11:49.713677Z"},"links":{"cited_paper":"/paper/2505.15879","citing_paper":"/paper/2607.05716"},"observation_digest":"sha256:6edf4ab621c226838402c0fc1782a1b165c31a35572b68e5f17fd37557f5815d","observation_id":"3f8d0cdb-a77f-47aa-a1d8-7b4799e5e5af","resolution":{"observed_at":"2026-07-11T03:17:51.534720Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.15879","snapshot_observed_at":"2026-07-14T16:12:58.882161Z","title":"J., Guan, X., and Wang, X","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.05716","last_updated":"2026-07-13T06:34:36Z","snapshot_observed_at":"2026-08-02T09:35:49.380987Z","submitted_at":"2026-07-07T01:00:51Z","title":"Scene Graph Thinking: Reinforcing Structured Visual Reasoning for Multimodal Large Language Models","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-07-14T16:12:58.882161Z"},"links":{"cited_paper":"/paper/2505.15879","citing_paper":"/paper/2607.05716"},"observation_digest":"sha256:066164878ded8f5a54ba77d814f3e2388d775ce6d43579c144c79621d486eca2","observation_id":"8c79a5ae-e61f-4544-b33e-d96d7766b491","resolution":{"observed_at":"2026-07-14T16:12:58.882161Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2505.15879/citation-record","integrity":"/paper/2505.15879/integrity","json":"/paper/2505.15879/citation-record.json","paper":"/paper/2505.15879"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Introducing openai o1-preview","venue":null,"work_id":"3744700e-1587-4536-a41b-a087aa4bf2ec","year":2024},"citing_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-22T13:29:47.529564Z"},"links":{"citing_paper":"/paper/2505.15879"},"observation_digest":"sha256:1a3d5d208bb50ac9d3597b365053b722e2223304bc2d6ea1accab966b8ff216a","observation_id":"53021e62-d29b-4050-b407-e80fd280bd56","resolution":{"observed_at":"2026-05-22T13:31:37.037775Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2501.12948","doi":"10.1016/j.artmed.2024.103001","metadata_source":"pith","pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","venue":"cs.CL","work_id":"e6b75ad5-2877-4168-97c8-710407094d20","year":2025},"citing_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-22T13:29:47.529564Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2505.15879"},"observation_digest":"sha256:10373aacb2512082dbe0df7dea14d1248b54f5d0979a6312a34ab4354d8165e4","observation_id":"63bce2bf-3454-4dab-9d1c-9c5edb01d2c5","resolution":{"observed_at":"2026-05-22T13:31:36.078942Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":"2412.15115","doi":"10.1145/3581783.3612503","metadata_source":"pith","pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Qwen2.5 Technical Report","venue":"cs.CL","work_id":"d8432992-4980-4a81-85c7-9fa2c2b87f85","year":2024},"citing_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-22T13:29:47.529564Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2505.15879"},"observation_digest":"sha256:93797eea7ea461e867d964a9b2d8629b63794199e6dace3894b1cc4e7c5b1ded","observation_id":"6074d927-0032-435b-b456-b504044db977","resolution":{"observed_at":"2026-05-22T13:31:36.045116Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Qwq-32b: Embracing the power of reinforcement learning, March 2025","venue":null,"work_id":"efe1fadf-fa77-4c13-8cc1-8a9fefda7883","year":2025},"citing_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-22T13:29:47.529564Z"},"links":{"citing_paper":"/paper/2505.15879"},"observation_digest":"sha256:13da596e1369934bd6ed69b7af373f60a08f5800525b8098cfe2dcaa6963eeb4","observation_id":"bbf9e3c2-9f00-4b18-8176-8d9ef96eb906","resolution":{"observed_at":"2026-05-22T13:31:36.985792Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-07-06T18:55:11.576666Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":"2407.21783","doi":"10.1016/s0749-0720(15","metadata_source":"pith","pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"The Llama 3 Herd of Models","venue":"cs.AI","work_id":"1549a635-88af-4ac1-acfe-51ae7bb53345","year":2024},"citing_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-22T13:29:47.529564Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2505.15879"},"observation_digest":"sha256:8ddbca1161d7d73bf9f570871bb955a2b217679e9d123411ecd6e594d5bbd48b","observation_id":"c391cfeb-9da5-46ce-a8c5-6f1e1022a39e","resolution":{"observed_at":"2026-05-22T13:31:36.133645Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T02:06:42.595164Z","title":"Chain-of-thought prompting elicits reasoning in large language models","venue":null,"work_id":"a3ac3b4d-8c58-4772-ad55-18e8fb162bd1","year":2022},"citing_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-22T13:29:47.529564Z"},"links":{"citing_paper":"/paper/2505.15879"},"observation_digest":"sha256:f110c77341be56ebe3fc32822030470d231bb867c6f593b93bc7acbb8898c225","observation_id":"62cf520a-1a8a-4664-88ca-11b6e63de363","resolution":{"observed_at":"2026-05-22T13:31:36.993621Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Reasoning models don’t always say what they think","venue":null,"work_id":"8f3c5be8-c5da-49bc-929a-32dcbec57cc6","year":2025},"citing_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-22T13:29:47.529564Z"},"links":{"citing_paper":"/paper/2505.15879"},"observation_digest":"sha256:31a7a502f1082a315e9f5658c5b06d0b7a4da848334a2088f0f25de9d9a908e6","observation_id":"ec736a2d-336a-4a22-9cae-61fe2c786e0b","resolution":{"observed_at":"2026-05-22T13:31:36.981347Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.12966","last_updated":"2023-10-13T02:41:28Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-24T17:59:17Z","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","version":3},"cited_work":{"arxiv_id":"2308.12966","doi":"10.48550/arxiv.2308.12966","metadata_source":"pith","pith_arxiv_id":"2308.12966","snapshot_observed_at":"2026-07-10T11:37:03.214945Z","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","venue":"cs.CV","work_id":"cbc2bb21-b6bb-46c0-80bf-107e195ffe10","year":2023},"citing_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-22T13:29:47.529564Z"},"links":{"cited_paper":"/paper/2308.12966","citing_paper":"/paper/2505.15879"},"observation_digest":"sha256:3ae3ed9fd6af59bd810f2f4b6d24f5bfe18604dc7b010ece53352d525db1791c","observation_id":"b2a39bc5-e1b9-4c1c-b816-3a72772522de","resolution":{"observed_at":"2026-05-22T13:31:36.095013Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.06769","last_updated":"2025-11-03T00:53:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-09T18:55:56Z","title":"Training Large Language Models to Reason in a Continuous Latent Space","version":3},"cited_work":{"arxiv_id":"2412.06769","doi":"10.48550/arxiv.2412.06769","metadata_source":"pith","pith_arxiv_id":"2412.06769","snapshot_observed_at":"2026-07-11T00:37:42.845448Z","title":"Training Large Language Models to Reason in a Continuous Latent Space","venue":"cs.CL","work_id":"3ddd0fd2-c176-408f-9b58-0666c2707f2d","year":2024},"citing_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-22T13:29:47.529564Z"},"links":{"cited_paper":"/paper/2412.06769","citing_paper":"/paper/2505.15879"},"observation_digest":"sha256:6b301e11560a8e725fa512b8466d3f0344d2433c96f7f34518790444878876e8","observation_id":"d578c0e4-75b8-4e2c-90be-49bfec98591d","resolution":{"observed_at":"2026-05-22T13:31:36.089157Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-05-23T16:25:44.826594+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-23T16:25:44.826594+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.07615","last_updated":"2025-04-14T15:15:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-10T10:05:15Z","title":"VLM-R1: A Stable and Generalizable R1-style Large Vision-Language Model","version":2},"cited_work":{"arxiv_id":"2504.07615","doi":null,"metadata_source":"pith","pith_arxiv_id":"2504.07615","snapshot_observed_at":"2026-07-11T03:17:51.904109Z","title":"VLM-R1: A Stable and Generalizable R1-style Large Vision-Language Model","venue":"cs.CV","work_id":"d36889cb-edb6-448f-9a50-36df8b1623e5","year":2025},"citing_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-22T13:29:47.529564Z"},"links":{"cited_paper":"/paper/2504.07615","citing_paper":"/paper/2505.15879"},"observation_digest":"sha256:74616e5770932fe7a8750de45fe3e25193c759e85e4f937fca06b74477dbdb64","observation_id":"3dfb7bd1-d7df-41ee-85bf-a5370b257c62","resolution":{"observed_at":"2026-05-22T13:31:36.050924Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06749","last_updated":"2026-02-28T21:10:52Z","snapshot_observed_at":"2026-07-06T20:49:27.466064Z","submitted_at":"2025-03-09T20:06:45Z","title":"Vision-R1: Incentivizing Reasoning Capability in Multimodal Large Language Models","version":4},"cited_work":{"arxiv_id":"2503.06749","doi":"10.48550/arxiv.2503.06749","metadata_source":"pith","pith_arxiv_id":"2503.06749","snapshot_observed_at":"2026-07-11T03:17:51.684746Z","title":"Vision-R1: Incentivizing Reasoning Capability in Multimodal Large Language Models","venue":"cs.CV","work_id":"38998646-34ee-4605-b661-ab356f16d6e5","year":2025},"citing_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-22T13:29:47.529564Z"},"links":{"cited_paper":"/paper/2503.06749","citing_paper":"/paper/2505.15879"},"observation_digest":"sha256:437887e79207967382ed12b2866ca60584bf0f10b6b7f8d8143850c43f0b4b1e","observation_id":"6856c958-98f0-4f8c-94a5-19c6668a4ff3","resolution":{"observed_at":"2026-05-22T13:31:36.084035Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10615","last_updated":"2025-03-18T08:52:34Z","snapshot_observed_at":"2026-07-06T20:52:09.743730Z","submitted_at":"2025-03-13T17:56:05Z","title":"R1-Onevision: Advancing Generalized Multimodal Reasoning through Cross-Modal Formalization","version":2},"cited_work":{"arxiv_id":"2503.10615","doi":"10.48550/arxiv.2503.10615","metadata_source":"pith","pith_arxiv_id":"2503.10615","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"R1-Onevision: Advancing Generalized Multimodal Reasoning through Cross-Modal Formalization","venue":"cs.CV","work_id":"bd2bf4d0-20bf-49b8-8dac-b54a8019be6c","year":2025},"citing_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-22T13:29:47.529564Z"},"links":{"cited_paper":"/paper/2503.10615","citing_paper":"/paper/2505.15879"},"observation_digest":"sha256:41dc941787ef9259006db1ce2b233c9fb00a65a0a13eacb1f44a9e583817ac60","observation_id":"fca9c436-30b5-427d-9a4c-14fb285614e1","resolution":{"observed_at":"2026-05-22T13:31:36.100958Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.10479","last_updated":"2025-04-19T03:47:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-14T17:59:25Z","title":"InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models","version":3},"cited_work":{"arxiv_id":"2504.10479","doi":"10.48550/arxiv.2504.10479","metadata_source":"pith","pith_arxiv_id":"2504.10479","snapshot_observed_at":"2026-07-11T03:17:51.831519Z","title":"InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models","venue":"cs.CV","work_id":"fe8637aa-12bc-4434-8d36-9f57b5eebcbe","year":2025},"citing_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-22T13:29:47.529564Z"},"links":{"cited_paper":"/paper/2504.10479","citing_paper":"/paper/2505.15879"},"observation_digest":"sha256:e021cbc1899f6f822b6e83ca6f9cb6eb4f2540750588dd08eaf0618bbf246ec5","observation_id":"0994b7bf-7d19-4bf0-8e6a-0ce65e4607dc","resolution":{"observed_at":"2026-05-22T13:31:36.139144Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-05-20T07:54:09.017512+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T07:54:09.017512+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Visual spatial reasoning","venue":null,"work_id":"983766d6-2519-440f-a679-c96e8b9e50f0","year":2023},"citing_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-22T13:29:47.529564Z"},"links":{"citing_paper":"/paper/2505.15879"},"observation_digest":"sha256:4bdbbdb13b234dcd41f944ae57f3c0e1d1f33c43608a369542e65522e1774d42","observation_id":"f8ee944a-7da6-49af-8e04-5d328533c36a","resolution":{"observed_at":"2026-05-22T13:31:36.997067Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Tallyqa: Answering complex counting questions","venue":null,"work_id":"3e4f24a6-5ad9-4d65-b005-9293391ca9a5","year":2019},"citing_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-22T13:29:47.529564Z"},"links":{"citing_paper":"/paper/2505.15879"},"observation_digest":"sha256:50992e131f9099e58a2fb66f56a2ad2cb1af5fa1f4ff9423c01e3d15d2e02db1","observation_id":"3fc15d87-0cb2-4b55-bbe2-70ea3dd3f003","resolution":{"observed_at":"2026-05-22T13:31:36.977947Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"R1-v: Reinforcing super generaliza- tion ability in vision-language models with less than $3.https://github.com/Deep-Agent/ R1-V","venue":null,"work_id":"f17dee65-ec0a-48ea-9169-842fa5973868","year":2025},"citing_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-22T13:29:47.529564Z"},"links":{"citing_paper":"/paper/2505.15879"},"observation_digest":"sha256:a478d8ba15d827b5b56705699e73dc5ca04e0fa6585ad938107b43aa27f0eac4","observation_id":"abd39a97-7d2f-4d16-98cf-d2c9f9c75a3d","resolution":{"observed_at":"2026-05-22T13:31:37.004702Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Introducing openai o3 and o4-mini","venue":null,"work_id":"7eac42cc-8650-4338-8afb-3a178010c37c","year":2025},"citing_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-22T13:29:47.529564Z"},"links":{"citing_paper":"/paper/2505.15879"},"observation_digest":"sha256:904bd6279e0fb3d399e15f2ddc5c8639e13a021810a9ab1edf3c388835b91fa7","observation_id":"5669a613-aa98-4d70-9955-d87363bd152e","resolution":{"observed_at":"2026-05-22T13:31:37.016889Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Self-consistency improves chain of thought reasoning in language models","venue":null,"work_id":"55eaa9bc-51b1-4aaa-8ee7-066ddb14d037","year":2022},"citing_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-22T13:29:47.529564Z"},"links":{"citing_paper":"/paper/2505.15879"},"observation_digest":"sha256:ad831d451b693717783b99f01e114402725b8516a10a2882359b08a5ee2d5e7e","observation_id":"3baa21e4-711b-4dad-800a-1daeb7ca1339","resolution":{"observed_at":"2026-05-22T13:31:37.024169Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.00923","last_updated":"2024-05-20T06:43:48Z","snapshot_observed_at":"2026-07-06T14:47:34.480641Z","submitted_at":"2023-02-02T07:51:19Z","title":"Multimodal Chain-of-Thought Reasoning in Language Models","version":5},"cited_work":{"arxiv_id":"2302.00923","doi":"10.48550/arxiv.2302.00923","metadata_source":"pith","pith_arxiv_id":"2302.00923","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Multimodal Chain-of-Thought Reasoning in Language Models","venue":"cs.CL","work_id":"cf983328-0c5e-4840-8f51-2b24cf4b21ee","year":2023},"citing_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-22T13:29:47.529564Z"},"links":{"cited_paper":"/paper/2302.00923","citing_paper":"/paper/2505.15879"},"observation_digest":"sha256:c275a672dfd8419afcfef986769e3c172078ed5f712040781e8b4ba8fce4f82e","observation_id":"1637b3ad-0836-4ee4-a4e2-097fff7616fc","resolution":{"observed_at":"2026-05-22T13:31:36.128814Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Visual chain-of-thought prompting for knowledge-based visual reasoning","venue":null,"work_id":"79609b54-3700-4a85-b11e-b7465034eae3","year":2024},"citing_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-22T13:29:47.529564Z"},"links":{"citing_paper":"/paper/2505.15879"},"observation_digest":"sha256:ae8414f2eda0925fb11f0cb0dc1a979558e59b944062563a29af58043a47e3a2","observation_id":"1f66ac9c-5bec-4b6f-a6a6-130f019ef756","resolution":{"observed_at":"2026-05-22T13:31:37.030672Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Compositional chain-of- thought prompting for large multimodal models","venue":null,"work_id":"439c7a1c-0a62-4bfa-98b3-0d5e3b98b3ee","year":2024},"citing_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-22T13:29:47.529564Z"},"links":{"citing_paper":"/paper/2505.15879"},"observation_digest":"sha256:f28362872496a87c83877e82354430c99439138c2d2009edf90babe018685f19","observation_id":"6ba154f8-3fae-4248-b5f6-8935ec9c9219","resolution":{"observed_at":"2026-05-22T13:31:37.027697Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.18397","last_updated":"2025-07-15T07:32:27Z","snapshot_observed_at":"2026-07-06T21:14:48.026751Z","submitted_at":"2025-04-25T14:48:18Z","title":"Unsupervised Visual Chain-of-Thought Reasoning via Preference Optimization","version":2},"cited_work":{"arxiv_id":"2504.18397","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2504.18397","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Unsupervised visual chain-of-thought reasoning via preference optimization","venue":null,"work_id":"b7dbcfa0-d7bc-4f47-b9f8-217f58cbc477","year":2025},"citing_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-22T13:29:47.529564Z"},"links":{"cited_paper":"/paper/2504.18397","citing_paper":"/paper/2505.15879"},"observation_digest":"sha256:2ccf13af0f67cbf22cb29466858a924721dea22aad6a6ffa2757a4f8c5717179","observation_id":"23d04d59-7f2a-437a-8a1e-806344182d1c","resolution":{"observed_at":"2026-05-22T13:31:36.146143Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Visual cot: Advancing multi-modal language models with a comprehen- sive dataset and benchmark for chain-of-thought reasoning","venue":null,"work_id":"5dc0a209-8920-4f9d-946d-67bbdff5ef44","year":2024},"citing_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-22T13:29:47.529564Z"},"links":{"citing_paper":"/paper/2505.15879"},"observation_digest":"sha256:e0b6e6e99041dd2c1ad84da92c7886759fd5ab3f0088938d94579c2a666f8466","observation_id":"25a769e3-2dd9-43f7-86f3-a0c6b7f535b2","resolution":{"observed_at":"2026-05-22T13:31:37.011643Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"cited_work":{"arxiv_id":"2402.03300","doi":"10.1016/0004-3702(73)90011-8","metadata_source":"pith","pith_arxiv_id":"2402.03300","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","venue":"cs.CL","work_id":"c5006563-f3ec-438a-9e35-b7b484f34828","year":2024},"citing_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-22T13:29:47.529564Z"},"links":{"cited_paper":"/paper/2402.03300","citing_paper":"/paper/2505.15879"},"observation_digest":"sha256:2b7ddc22c9dded7823b1dc63a8bcb0adcf9bb173327391ab4835f44ea960b06c","observation_id":"04ded7dc-4405-40b8-a22d-83dbb6eb014a","resolution":{"observed_at":"2026-05-22T13:31:36.068306Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T15:02:40.822401Z","title":"Gqa: A new dataset for real-world visual reasoning and compositional question answering","venue":null,"work_id":"e114d408-7f00-41ea-98b0-9e628094fec0","year":2019},"citing_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-22T13:29:47.529564Z"},"links":{"citing_paper":"/paper/2505.15879"},"observation_digest":"sha256:2fde991416acb1c84309ff2b3cefd25ca28eff8e36de86d189a7d8a6f9248f6e","observation_id":"57b627d1-387d-4273-a040-2f30f4199571","resolution":{"observed_at":"2026-05-22T13:31:37.007984Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Mme: A comprehensive evaluation benchmark for multimodal large language models","venue":null,"work_id":"168ec936-9cd1-41ee-a142-ab1dae9b73a4","year":2024},"citing_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-22T13:29:47.529564Z"},"links":{"citing_paper":"/paper/2505.15879"},"observation_digest":"sha256:f5c94d64a92da5dbc731b7c88fcf9d3c50c336f5fc4b4666e17d6f1833c732d6","observation_id":"5e7b5baf-aae9-4a4d-a81f-9470be0b2def","resolution":{"observed_at":"2026-05-22T13:31:37.034155Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02255","last_updated":"2024-01-21T03:47:06Z","snapshot_observed_at":"2026-07-06T16:27:15.027202Z","submitted_at":"2023-10-03T17:57:24Z","title":"MathVista: Evaluating Mathematical Reasoning of Foundation Models in Visual Contexts","version":3},"cited_work":{"arxiv_id":"2310.02255","doi":"10.1109/cvpr52734.2025.01245","metadata_source":"pith","pith_arxiv_id":"2310.02255","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"MathVista: Evaluating Mathematical Reasoning of Foundation Models in Visual Contexts","venue":"cs.CV","work_id":"e22c3789-9e71-4242-b6ea-3e60e06e2b66","year":2023},"citing_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-22T13:29:47.529564Z"},"links":{"cited_paper":"/paper/2310.02255","citing_paper":"/paper/2505.15879"},"observation_digest":"sha256:e4c8eaf1ca4f1248540ee9eb11ffe3ed924e2660c1b1d6a3d07624e949873ffc","observation_id":"497d4ed4-be8a-4635-ae1a-43e62397c992","resolution":{"observed_at":"2026-05-22T13:31:36.057778Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.13177","last_updated":"2023-12-18T07:29:55Z","snapshot_observed_at":"2026-08-01T19:22:07.600105Z","submitted_at":"2023-08-25T04:54:32Z","title":"How to Evaluate the Generalization of Detection? A Benchmark for Comprehensive Open-Vocabulary Detection","version":2},"cited_work":{"arxiv_id":"2308.13177","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2308.13177","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"How to evaluate the generalization of detection? a benchmark for comprehensive open-vocabulary detection","venue":null,"work_id":"4359f02a-7fae-41b9-a417-541523bfefa6","year":2023},"citing_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-22T13:29:47.529564Z"},"links":{"cited_paper":"/paper/2308.13177","citing_paper":"/paper/2505.15879"},"observation_digest":"sha256:8abf33c40555f181aff80563152ac92c12ed7b712f5c6a50c30a0fac45d67334","observation_id":"45712bc3-ddea-46cb-8d18-992d0bc06afc","resolution":{"observed_at":"2026-05-22T13:31:36.117238Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Judging llm-as-a-judge with mt-bench and chatbot arena","venue":null,"work_id":"e2f91466-a8ba-4238-8142-7801e553fa10","year":2023},"citing_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-22T13:29:47.529564Z"},"links":{"citing_paper":"/paper/2505.15879"},"observation_digest":"sha256:a159b7bff380d2cf32d60930f788129a07df9d7f39d6a90f2b09a5be048c30fb","observation_id":"d61dcef2-1214-43f4-b902-98e7ea7f22bf","resolution":{"observed_at":"2026-05-22T13:31:37.001013Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2005.14165","last_updated":"2020-07-22T19:47:17Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-05-28T17:29:03Z","title":"Language Models are Few-Shot Learners","version":4},"cited_work":{"arxiv_id":"2005.14165","doi":"10.1145/1926385.1926423","metadata_source":"pith","pith_arxiv_id":"2005.14165","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Language Models are Few-Shot Learners","venue":"cs.CL","work_id":"214732c0-2edd-44a0-af9e-28184a2b8279","year":2020},"citing_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-22T13:29:47.529564Z"},"links":{"cited_paper":"/paper/2005.14165","citing_paper":"/paper/2505.15879"},"observation_digest":"sha256:762b4ba107f71cfc5cb20dacc5c961108e13dcd93d13f49c6658653311fb8cea","observation_id":"e8c31818-4823-4ef0-8fe6-ee62462f9c88","resolution":{"observed_at":"2026-05-22T13:31:36.062621Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-07-09T08:48:33.42086+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-09T08:48:33.42086+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.11441","last_updated":"2023-11-06T07:39:49Z","snapshot_observed_at":"2026-08-04T03:29:49.409446Z","submitted_at":"2023-10-17T17:51:31Z","title":"Set-of-Mark Prompting Unleashes Extraordinary Visual Grounding in GPT-4V","version":2},"cited_work":{"arxiv_id":"2310.11441","doi":"10.48550/arxiv.2310.11441","metadata_source":"pith","pith_arxiv_id":"2310.11441","snapshot_observed_at":"2026-07-10T23:17:45.053170Z","title":"Set-of-Mark Prompting Unleashes Extraordinary Visual Grounding in GPT-4V","venue":"cs.CV","work_id":"ddc8878e-ed0c-4a66-952a-254dce1c622a","year":2023},"citing_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-22T13:29:47.529564Z"},"links":{"cited_paper":"/paper/2310.11441","citing_paper":"/paper/2505.15879"},"observation_digest":"sha256:67c5c4ee78bfeaeaeffcd3f749ccfc0e1db8029165bef1efcf3273298a57401b","observation_id":"f1440bbd-83fc-4df3-9e49-8a0c4355dd24","resolution":{"observed_at":"2026-05-22T13:31:36.106717Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.13837","last_updated":"2025-11-24T06:11:04Z","snapshot_observed_at":"2026-07-06T21:11:34.701779Z","submitted_at":"2025-04-18T17:59:56Z","title":"Does Reinforcement Learning Really Incentivize Reasoning Capacity in LLMs Beyond the Base Model?","version":5},"cited_work":{"arxiv_id":"2504.13837","doi":"10.48550/arxiv.2504.13837","metadata_source":"pith","pith_arxiv_id":"2504.13837","snapshot_observed_at":"2026-07-10T22:47:36.894457Z","title":"Does Reinforcement Learning Really Incentivize Reasoning Capacity in LLMs Beyond the Base Model?","venue":"cs.AI","work_id":"d854765a-e664-41c0-8655-21c4bf2e0cc4","year":2025},"citing_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-22T13:29:47.529564Z"},"links":{"cited_paper":"/paper/2504.13837","citing_paper":"/paper/2505.15879"},"observation_digest":"sha256:b628fcee5119154923f4cc304b8f402787b809bfa30d33af705babc6c981ee22","observation_id":"bf26a524-4fa1-4caa-ae18-5990dab4cdd4","resolution":{"observed_at":"2026-05-22T13:31:36.073266Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-07-09T10:48:42.002839+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-09T10:48:42.002839+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.17352","last_updated":"2025-11-11T08:13:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-21T17:52:43Z","title":"OpenVLThinker: Complex Vision-Language Reasoning via Iterative SFT-RL Cycles","version":3},"cited_work":{"arxiv_id":"2503.17352","doi":"10.48550/arxiv.2503.17352","metadata_source":"pith","pith_arxiv_id":"2503.17352","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"OpenVLThinker: Complex Vision-Language Reasoning via Iterative SFT-RL Cycles","venue":"cs.CV","work_id":"de4c64e1-82b1-4f70-8311-a3539e7bf400","year":2025},"citing_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-22T13:29:47.529564Z"},"links":{"cited_paper":"/paper/2503.17352","citing_paper":"/paper/2505.15879"},"observation_digest":"sha256:87960d0a2746f1584ce73449ed433c4e5537f37d9593cd243754a6f8a2633d9c","observation_id":"4ef8c58e-78f1-4d42-ad0b-608a17815f64","resolution":{"observed_at":"2026-05-22T13:31:36.111936Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.11468","last_updated":"2025-04-10T16:54:05Z","snapshot_observed_at":"2026-07-06T21:09:55.394184Z","submitted_at":"2025-04-10T16:54:05Z","title":"SFT or RL? An Early Investigation into Training R1-Like Reasoning Large Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2504.11468","doi":"10.48550/arxiv.2504.11468","metadata_source":"pith","pith_arxiv_id":"2504.11468","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"SFT or RL? An Early Investigation into Training R1-Like Reasoning Large Vision-Language Models","venue":"cs.CL","work_id":"a521360c-8673-4d0d-a3a3-6eb9f7a71b90","year":2025},"citing_paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-22T13:29:47.529564Z"},"links":{"cited_paper":"/paper/2504.11468","citing_paper":"/paper/2505.15879"},"observation_digest":"sha256:f8163563d7638d4e6762223be1a58401d7c6e0414eaf1c938fa89346e945f902","observation_id":"6388e0e0-b733-4138-9ed3-b930cdf92b31","resolution":{"observed_at":"2026-05-22T13:31:36.123124Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-04T06:34:03.388597+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2505.15879","last_updated":"2026-05-09T18:01:18Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-21T17:54:49Z","title":"GRIT: Teaching MLLMs to Think with Images"},"reference_resolution":{"displayed":34,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":0,"verified_exact":18,"verified_fuzzy":15},"total_outbound_references":34},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-04T06:34:03.388597+00:00","source":"crossref"},{"observed_at":"2026-08-04T06:33:57.428241+00:00","source":"retraction_watch"}],"thesis":"As of 4 August 2026, this Paper Citation Record lists 34 of 34 outbound references and 34 inbound Pith citation observations for arXiv:2505.15879."}