{"as_of":"2026-08-02T12:27:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:96e01bf97b83f0fae2aba70e3eda24c27832963df5bb34915ff0223ffbc18a29","coverage":[{"denominator":40,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":40,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-13T23:57:47.657243Z","state":"measured"},{"denominator":43,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":43,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-02T06:30:47.504484+00:00","state":"measured"},{"denominator":3,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":3,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-13T06:45:27.857034Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-09T13:56:19.173208Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"cited_work":{"arxiv_id":"2604.03307","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.03307","snapshot_observed_at":"2026-07-09T13:56:19.173208Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","venue":"cs.CV","work_id":"81eff3fe-39a1-456d-be35-5e6a875a5cca","year":2026},"citing_paper":{"arxiv_id":"2606.00562","last_updated":"2026-05-30T06:33:24Z","snapshot_observed_at":"2026-07-06T23:41:15.410587Z","submitted_at":"2026-05-30T06:33:24Z","title":"DeepLatent: Think with Images via Parallel Latent Visual Reasoning","version":1},"reference_index":86,"source":"arxiv_source","source_observed_at":"2026-06-28T18:44:39.545911Z"},"links":{"cited_paper":"/paper/2604.03307","citing_paper":"/paper/2606.00562"},"observation_digest":"sha256:b5846574d8d1c4b50a87eb7383d66b6d3bd61b903f29794dbe75c64a2aeaabfa","observation_id":"d9cbee03-fca8-467c-b2e4-33c98b9a092a","resolution":{"observed_at":"2026-06-28T20:22:37.724913Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"cited_work":{"arxiv_id":"2604.03307","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.03307","snapshot_observed_at":"2026-07-09T13:56:19.173208Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","venue":"cs.CV","work_id":"81eff3fe-39a1-456d-be35-5e6a875a5cca","year":2026},"citing_paper":{"arxiv_id":"2607.07361","last_updated":"2026-07-10T08:24:37Z","snapshot_observed_at":"2026-08-02T01:45:49.287044Z","submitted_at":"2026-07-08T12:56:39Z","title":"BUS: Brain-Inspired Unsupervised Self-Reflection via Backward Prediction for Multimodal Reasoning","version":1},"reference_index":98,"source":"arxiv_source","source_observed_at":"2026-07-09T13:51:49.149342Z"},"links":{"cited_paper":"/paper/2604.03307","citing_paper":"/paper/2607.07361"},"observation_digest":"sha256:b8f19dc743a21353f0f0da87d121000489c384f94dbb0f92d3ece12f03e230ad","observation_id":"c2ddf2ac-a0d9-4960-9452-93d250756ea7","resolution":{"observed_at":"2026-07-09T13:56:19.174675Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2604.03307","snapshot_observed_at":"2026-07-13T06:45:27.857034Z","title":"arXiv preprint arXiv:2604.03307 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.07361","last_updated":"2026-07-10T08:24:37Z","snapshot_observed_at":"2026-08-02T01:45:49.287044Z","submitted_at":"2026-07-08T12:56:39Z","title":"BUS: Brain-Inspired Unsupervised Self-Reflection via Backward Prediction for Multimodal Reasoning","version":2},"reference_index":98,"source":"arxiv_source","source_observed_at":"2026-07-13T06:45:27.857034Z"},"links":{"cited_paper":"/paper/2604.03307","citing_paper":"/paper/2607.07361"},"observation_digest":"sha256:a43fc1dbdc13ff00c0a5c76a45163ed750c73066b3357f6224cfe1b6c5fcde0e","observation_id":"0276d545-cc89-4624-9e14-bdd0a3a33304","resolution":{"observed_at":"2026-07-13T06:45:27.857034Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2604.03307/citation-record","integrity":"/paper/2604.03307/integrity","json":"/paper/2604.03307/citation-record.json","paper":"/paper/2604.03307"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2309.16609","last_updated":"2023-09-28T17:07:49Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-09-28T17:07:49Z","title":"Qwen Technical Report","version":1},"cited_work":{"arxiv_id":"2309.16609","doi":"10.48550/arxiv.2309.16609","metadata_source":"pith","pith_arxiv_id":"2309.16609","snapshot_observed_at":"2026-07-10T20:57:35.014119Z","title":"Qwen Technical Report","venue":"cs.CL","work_id":"bb1fd52f-6b2f-437c-9516-37bdf6eb9be8","year":2023},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2309.16609","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:b0b1b29d2ad61b7a1418975fba1f2ec0f8d0cb8c6fbe392d28d5304cf6d83157","observation_id":"56382967-f85d-4383-bfc7-4f1fdcae5cc3","resolution":{"observed_at":"2026-05-13T23:58:28.490776Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-07-15T23:50:15.620681+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-15T23:50:15.620681+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2511.21631","last_updated":"2025-11-27T12:16:54Z","snapshot_observed_at":"2026-07-06T22:37:03.716474Z","submitted_at":"2025-11-26T17:59:08Z","title":"Qwen3-VL Technical Report","version":2},"cited_work":{"arxiv_id":"2511.21631","doi":"10.1016/j.neunet.2025.107777","metadata_source":"pith","pith_arxiv_id":"2511.21631","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Qwen3-VL Technical Report","venue":"cs.CV","work_id":"1fe243aa-e3c0-4da6-b391-4cbcfc88d5c0","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2511.21631","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:7702d302d908197044ae098013f2672cae3aa5076cd96e31e984cdee61555322","observation_id":"bd5e5be2-8470-4c5c-822d-3b59970544e4","resolution":{"observed_at":"2026-05-13T23:58:28.621212Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.05271","last_updated":"2025-09-26T12:52:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-06T18:57:08Z","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","version":5},"cited_work":{"arxiv_id":"2412.05271","doi":"10.48550/arxiv.2412.05271","metadata_source":"pith","pith_arxiv_id":"2412.05271","snapshot_observed_at":"2026-07-11T01:17:44.298728Z","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","venue":"cs.CV","work_id":"ee70bdc8-4656-4849-ada7-ce42a2278d70","year":2024},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2412.05271","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:3274f4c56509e055af2748d3358a518dc4a41272d7999aa0ddbf443f158a9864","observation_id":"9da4abac-eacf-4abb-8323-84cf38282d8c","resolution":{"observed_at":"2026-05-13T23:58:28.519794Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-07-11T02:19:07.858539+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T02:19:07.858539+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Internvl: Scaling up vision foundation models and aligning for generic 11 visual-linguistic tasks","venue":null,"work_id":"44467007-69a1-4604-97f9-3f125b378f31","year":2024},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:a573ed5d51b30013afa19be9f6b4b2e0c3d21c203510c6b2c48dcc37bbac96f0","observation_id":"2c1c0388-0b5c-4c06-b305-3f1797da3633","resolution":{"observed_at":"2026-05-13T23:58:29.519461Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.13171","last_updated":"2024-12-17T18:50:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-17T18:50:33Z","title":"Compressed Chain of Thought: Efficient Reasoning Through Dense Representations","version":1},"cited_work":{"arxiv_id":"2412.13171","doi":"10.48550/arxiv.2412.13171","metadata_source":"pith","pith_arxiv_id":"2412.13171","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Compressed Chain of Thought: Efficient Reasoning Through Dense Representations","venue":"cs.CL","work_id":"5d72fcbb-d14d-4ac0-8644-50807a64d543","year":2024},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2412.13171","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:32fd3a6c83bc4c7d6f2ea07fd92c874a5e423a65180ff78420b55daf938fc0de","observation_id":"e77e0b81-a7ff-485c-a4ad-ea6dc231f864","resolution":{"observed_at":"2026-05-17T04:47:40.470022Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.14683","last_updated":"2025-07-27T11:45:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-20T17:59:30Z","title":"Emerging Properties in Unified Multimodal Pretraining","version":3},"cited_work":{"arxiv_id":"2505.14683","doi":"10.48550/arxiv.2505.14683","metadata_source":"pith","pith_arxiv_id":"2505.14683","snapshot_observed_at":"2026-07-10T13:37:06.841769Z","title":"Emerging Properties in Unified Multimodal Pretraining","venue":"cs.CV","work_id":"e0cfd82c-f5d4-44fd-b531-ec73ab0a805b","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2505.14683","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:7b5a8cd39d30ea79394c9c3a3ec19205bcf3f10d4f1e9e8fceb9df50db4177e6","observation_id":"605c65a9-2d20-43ba-beb3-2ee3961beb76","resolution":{"observed_at":"2026-05-13T23:58:28.578934Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Blink: Multimodal large language models can see but not perceive","venue":null,"work_id":"db9efe85-b205-4b6c-897b-aca7d1c3321a","year":2024},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:3576df444e08069bf57d2d938a7a1da97cc0cdb2b267d7ee22682917a9f06c93","observation_id":"6a5d5814-3ba6-443a-8b76-c8c769a3f8f9","resolution":{"observed_at":"2026-05-13T23:58:29.507759Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.06769","last_updated":"2025-11-03T00:53:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-09T18:55:56Z","title":"Training Large Language Models to Reason in a Continuous Latent Space","version":3},"cited_work":{"arxiv_id":"2412.06769","doi":"10.48550/arxiv.2412.06769","metadata_source":"pith","pith_arxiv_id":"2412.06769","snapshot_observed_at":"2026-07-11T00:37:42.845448Z","title":"Training Large Language Models to Reason in a Continuous Latent Space","venue":"cs.CL","work_id":"3ddd0fd2-c176-408f-9b58-0666c2707f2d","year":2024},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2412.06769","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:70dc1feb8c9ed6620715b52cebd9ca697457efc146926cab7cf17ec7c0b3908e","observation_id":"f54922e4-55e5-4b2a-ace6-55795e4ea7d0","resolution":{"observed_at":"2026-05-13T23:58:28.634771Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-05-23T16:25:44.826594+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-23T16:25:44.826594+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06749","last_updated":"2026-02-28T21:10:52Z","snapshot_observed_at":"2026-07-06T20:49:27.466064Z","submitted_at":"2025-03-09T20:06:45Z","title":"Vision-R1: Incentivizing Reasoning Capability in Multimodal Large Language Models","version":4},"cited_work":{"arxiv_id":"2503.06749","doi":"10.48550/arxiv.2503.06749","metadata_source":"pith","pith_arxiv_id":"2503.06749","snapshot_observed_at":"2026-07-11T03:17:51.684746Z","title":"Vision-R1: Incentivizing Reasoning Capability in Multimodal Large Language Models","venue":"cs.CV","work_id":"38998646-34ee-4605-b661-ab356f16d6e5","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2503.06749","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:3aa4a4e29337fca92328ab46c5048727c4fb7f0f75c7d94c3f048bdd1ed26780","observation_id":"5aef65bd-63e6-4570-85c2-1007937c2f20","resolution":{"observed_at":"2026-05-13T23:58:28.536112Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21276","last_updated":"2024-10-25T17:43:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-25T17:43:01Z","title":"GPT-4o System Card","version":1},"cited_work":{"arxiv_id":"2410.21276","doi":"10.1177/15248380231178756","metadata_source":"pith","pith_arxiv_id":"2410.21276","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"GPT-4o System Card","venue":"cs.CL","work_id":"f37bf1c7-4964-4e56-9762-d20da8d9009f","year":2024},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2410.21276","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:40499a3183275149ad4d0ec3eb24f758116d4c8acfd23de8d2abe8cebf7f3ab1","observation_id":"97d476b3-748c-4412-8e97-f641cc46561a","resolution":{"observed_at":"2026-05-13T23:58:28.430718Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.04034","last_updated":"2025-06-04T14:56:57Z","snapshot_observed_at":"2026-07-06T21:36:35.142284Z","submitted_at":"2025-06-04T14:56:57Z","title":"Rex-Thinker: Grounded Object Referring via Chain-of-Thought Reasoning","version":1},"cited_work":{"arxiv_id":"2506.04034","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.04034","snapshot_observed_at":"2026-07-10T05:46:50.382039Z","title":"Rex-thinker: Grounded object re- ferring via chain-of-thought reasoning","venue":"cs.CV","work_id":"8f7606c1-62bd-42f7-8c1d-f545ea62a1d6","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2506.04034","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:a054bba215097d14883579610506a3cd3631f543b641982670658e94448c8ef2","observation_id":"1ef4b339-5566-48bd-bcdd-e477fa0ed9ea","resolution":{"observed_at":"2026-05-13T23:58:28.440054Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.24251","last_updated":"2025-10-05T04:01:18Z","snapshot_observed_at":"2026-07-06T22:31:00.843452Z","submitted_at":"2025-09-29T03:52:01Z","title":"Latent Visual Reasoning","version":2},"cited_work":{"arxiv_id":"2509.24251","doi":"10.48550/arxiv.2509.24251","metadata_source":"pith","pith_arxiv_id":"2509.24251","snapshot_observed_at":"2026-07-11T03:17:51.433935Z","title":"Latent Visual Reasoning","venue":"cs.CV","work_id":"b6468cfa-4f13-4e02-b0ec-24ff5cd6785a","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2509.24251","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:6210e01f7278c91497b1349adf772077b1ac5127f98e4250a97b715df16dc3cc","observation_id":"1fe3b71d-7ac3-4323-867d-fa0b304081b6","resolution":{"observed_at":"2026-05-15T18:41:30.500607Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03326","last_updated":"2024-10-26T16:35:13Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-06T17:59:44Z","title":"LLaVA-OneVision: Easy Visual Task Transfer","version":3},"cited_work":{"arxiv_id":"2408.03326","doi":"10.48550/arxiv.2408.03326","metadata_source":"pith","pith_arxiv_id":"2408.03326","snapshot_observed_at":"2026-07-10T13:27:05.574543Z","title":"LLaVA-OneVision: Easy Visual Task Transfer","venue":"cs.CV","work_id":"f5f2452b-f2a9-49ac-b38d-c76e18cdfe49","year":2024},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2408.03326","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:b2f75a3aa09ba82036fe894298e95a65411850e50c34821945a71f9b7571bc9a","observation_id":"59493c6e-3948-4dd4-99fc-9ad9a9619ef9","resolution":{"observed_at":"2026-05-13T23:58:28.627283Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Visual-rft: Visual reinforcement fine-tuning","venue":null,"work_id":"97f7b2b7-ce37-410e-a822-be2d5f9a52ea","year":2034},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:baf28a7f33c4584895e23a5174c77e49a804bc6e1b8a3a81161ad7e0b673192d","observation_id":"c7782cb0-5b4e-40a7-b348-15a634e7e941","resolution":{"observed_at":"2026-05-13T23:58:29.576437Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.19702","last_updated":"2025-05-26T08:54:14Z","snapshot_observed_at":"2026-07-06T21:30:28.303379Z","submitted_at":"2025-05-26T08:54:14Z","title":"Point-RFT: Improving Multimodal Reasoning with Visually Grounded Reinforcement Finetuning","version":1},"cited_work":{"arxiv_id":"2505.19702","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.19702","snapshot_observed_at":"2026-07-04T19:50:09.831299Z","title":"Point-rft: Improving multimodal reasoning with visually grounded reinforcement finetuning.arXiv preprint arXiv:2505.19702","venue":null,"work_id":"34322379-c097-4ea5-89b9-7340f08e19b2","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2505.19702","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:81574d440d2c789d92b0579945d7bde2f6de241acd2701afce025b860b30919c","observation_id":"8e5f7f26-dd15-4346-9226-aca42d9c3d33","resolution":{"observed_at":"2026-05-13T23:58:28.614087Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.07536","last_updated":"2025-03-11T03:32:59Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-10T17:04:14Z","title":"LMM-R1: Empowering 3B LMMs with Strong Reasoning Abilities Through Two-Stage Rule-Based RL","version":2},"cited_work":{"arxiv_id":"2503.07536","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.07536","snapshot_observed_at":"2026-07-04T10:39:45.355868Z","title":"LMM-R1: Empowering 3B LMMs with Strong Reasoning Abilities Through Two-Stage Rule-Based RL","venue":"cs.CL","work_id":"30c18d3e-432d-404b-9572-1c7375bee8ed","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2503.07536","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:dc4228a6dfa8bce0b8fb59cacf4f2ebaf01712ddcc736b88f2353ef7d5759e56","observation_id":"4ae5fe2d-3889-4424-8438-20a3013dad52","resolution":{"observed_at":"2026-05-16T15:15:46.589538Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2511.19418","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T13:37:06.883595Z","title":"Chain-of-visual-thought: Teaching vlms to see and think better with continuous visual tokens","venue":null,"work_id":"76c55114-86cf-4cf0-b444-3806dfe547a0","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:f3d03cb279860cd3e79e0309c45118d5f3fb5a24bcb42c8badc5b34830454df9","observation_id":"fe43b6ab-f6b0-41d2-9969-3d02a9f9dab1","resolution":{"observed_at":"2026-05-13T23:58:28.514256Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"54ad3774-370f-4089-9684-4f33e9263602","year":2024},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:4f68f6368b0f6d92d7fc2102a2ee9f5a8d30c3ff6a98398311586e774c53f1b5","observation_id":"498e3f9d-8680-4b54-a970-898071c42b21","resolution":{"observed_at":"2026-05-13T23:58:29.556548Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Codi: Compressing chain-of-thought into continuous space via self-distillation","venue":null,"work_id":"c78e8405-f8ba-4d9d-bbc9-d76bd5f12b14","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:1e5832baf755a0185926bd500d04e6f707923eac96848fa3992bbff0f4c0be01","observation_id":"fecd0122-ba95-4896-9037-0a37a5c66983","resolution":{"observed_at":"2026-05-13T23:58:29.561411Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.08617","last_updated":"2025-07-09T14:53:06Z","snapshot_observed_at":"2026-07-06T21:23:19.117703Z","submitted_at":"2025-05-13T14:35:51Z","title":"OpenThinkIMG: Learning to Think with Images via Visual Tool Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2505.08617","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.08617","snapshot_observed_at":"2026-07-04T20:10:07.295499Z","title":"OpenThinkIMG: Learning to Think with Images via Visual Tool Reinforcement Learning","venue":"cs.CV","work_id":"3939edf9-d5d5-4c79-bd9c-02e7961fda21","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2505.08617","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:d91176fdb95e17c946745719d97685c2336311087e34af761885a5e1432fa8ba","observation_id":"cd2c0691-27e4-4e05-9179-be2baea0d56a","resolution":{"observed_at":"2026-05-16T22:13:13.788897Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Reason-rft: Reinforcement fine-tuning for visual reasoning.arXiv e-prints, pages arXiv–2503","venue":null,"work_id":"823b0c26-9d6f-4d77-ac65-df893f527561","year":null},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:f9c5661266b18bd04d9a6aa39a16c3a9f53386836f8946db48c536883bc7c9d7","observation_id":"005aa765-86f9-47e7-9a9d-25355ef293c7","resolution":{"observed_at":"2026-05-13T23:58:29.567637Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Eyes wide shut? exploring the visual shortcomings of multimodal llms","venue":null,"work_id":"e8b9cdb1-443e-4aad-b9d7-dcedd0d15f98","year":2024},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:8df99b2869f8fc83aa2e22b13e46dbbd6cff08c207ad52fabafe540a2318c885","observation_id":"4d3a1248-6fc4-4d6b-b1a0-49d4f7afe4dd","resolution":{"observed_at":"2026-05-13T23:58:29.515196Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15966","last_updated":"2025-10-24T09:35:22Z","snapshot_observed_at":"2026-07-06T21:28:06.120576Z","submitted_at":"2025-05-21T19:35:08Z","title":"Pixel Reasoner: Incentivizing Pixel-Space Reasoning with Curiosity-Driven Reinforcement Learning","version":3},"cited_work":{"arxiv_id":"2505.15966","doi":"10.48550/arxiv.2505.15966","metadata_source":"pith","pith_arxiv_id":"2505.15966","snapshot_observed_at":"2026-07-10T13:37:06.857138Z","title":"Pixel Reasoner: Incentivizing Pixel-Space Reasoning with Curiosity-Driven Reinforcement Learning","venue":"cs.CV","work_id":"878c3e90-ce55-4ba3-a588-2abe369013e6","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2505.15966","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:ca471a0d47336d3408e2a730f7c3e17c668e44828a6b4230d3ce234d4acfc69c","observation_id":"b4775fcf-8cfb-4ec1-b317-b1c75584d709","resolution":{"observed_at":"2026-05-14T02:22:27.090901Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2511.21395","doi":"10.48550/arxiv.2511.21395","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"Monet: Reasoning in latent visual space beyond images and language","venue":null,"work_id":"ce4ac531-7907-4d8a-afe2-9a0dae21ffa5","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:3466e32acc00c319d3a7f27a0b6652d24f181344bf359208e2cc09fc07539cd5","observation_id":"c06d79e1-8957-4c62-a998-f2bd30d29bfd","resolution":{"observed_at":"2026-05-13T23:58:28.543353Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.18265","last_updated":"2025-08-27T14:39:45Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-08-25T17:58:17Z","title":"InternVL3.5: Advancing Open-Source Multimodal Models in Versatility, Reasoning, and Efficiency","version":2},"cited_work":{"arxiv_id":"2508.18265","doi":"10.48550/arxiv.2508.18265","metadata_source":"pith","pith_arxiv_id":"2508.18265","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"InternVL3.5: Advancing Open-Source Multimodal Models in Versatility, Reasoning, and Efficiency","venue":"cs.CV","work_id":"b8f5e260-fff5-444e-bcf5-2c42cfefd83d","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2508.18265","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:3cc308e2230028a4820124f9fb388439500246f7bb21d47969df9360e99f127c","observation_id":"6285fb97-d596-4582-9ab9-a50195a27335","resolution":{"observed_at":"2026-05-13T23:58:28.600076Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Divide, conquer and combine: A training-free framework for high-resolution image perception in multimodal large language models","venue":null,"work_id":"66e104e2-61b5-4280-874c-9a724db39fde","year":2024},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:d501cf2ccada353ecc8975ed4b82d5067da82d5ea41e4b115c40a31a6acf7281","observation_id":"1a511570-42be-4fb4-9955-2269a7bc8953","resolution":{"observed_at":"2026-05-13T23:58:29.544943Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.06448","last_updated":"2026-04-14T16:31:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-08T23:22:34Z","title":"Perception-Aware Policy Optimization for Multimodal Reasoning","version":5},"cited_work":{"arxiv_id":"2507.06448","doi":"10.48550/arxiv.2507.06448","metadata_source":"pith","pith_arxiv_id":"2507.06448","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"Perception-Aware Policy Optimization for Multimodal Reasoning","venue":"cs.CL","work_id":"21674beb-d5af-4cf7-a1e0-c994ecedde54","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2507.06448","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:f0ea04de7f0a19741105261624cfb8fa6c58b294866a60a6de72d3d5bcfdbba7","observation_id":"64d0fd5c-6c9a-43a2-8266-583407ae2783","resolution":{"observed_at":"2026-05-13T23:58:28.528420Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Chain-of-thought prompting elicits reasoning in large language models.Advances in neural information processing systems, 35:24824–24837","venue":null,"work_id":"39958e49-4f07-4e4b-9a73-35c9b8bb3343","year":2022},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:31cf34b5f4f76c41176dd1814b5fb50b90fda3a4a4fafca50cf29efe1504a28b","observation_id":"b2c3769f-d601-4f3d-8b57-f678d12c6beb","resolution":{"observed_at":"2026-05-13T23:58:29.549587Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2505.19255","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T10:09:44.987304Z","title":"Vtool-r1: Vlms learn to think with images via reinforcement learning on multimodal tool use","venue":null,"work_id":"09e9aeff-7314-468c-b3c4-e45fe4def91a","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:371454215096ebc21119e631f7143acc5abb40f8708fe22960a456eb0b320108","observation_id":"aa29c6f0-164e-43a3-af49-7f78e3ef4753","resolution":{"observed_at":"2026-05-13T23:58:28.593993Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"V?: Guided visual search as a core mechanism in multimodal llms","venue":null,"work_id":"3637a7ac-83c6-48e2-b42b-39e251f6733e","year":2024},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:c015fb892ea37a3c0d16d4b89bba2f9746ccddb95a7d8be2ea32b87794969850","observation_id":"4a31d6bd-3e61-4c0b-abaf-38c030a5807a","resolution":{"observed_at":"2026-05-13T23:58:29.533713Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Llava-cot: Let vision language models reason step-by-step","venue":null,"work_id":"200c68d1-c21b-440a-ba0b-db6c8970af2c","year":2087},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:6f77499173882d68b28de672aaa58aaf6a9fcde60ee1c4329100b9ccc29be3f4","observation_id":"54428619-bef9-4b63-b9af-7efeee0f72b9","resolution":{"observed_at":"2026-05-13T23:58:29.538739Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Mc-bench: A benchmark for multi-context visual grounding in the era of mllms","venue":null,"work_id":"623c4bfe-802b-416c-98ba-f9a1945b519f","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:439a59a2596afc67bd1964c389e004e28d26398d37a0abd605d5c70f97004b54","observation_id":"2640e6fe-3f95-44d8-9628-0d999270d5c9","resolution":{"observed_at":"2026-05-13T23:58:29.523946Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"R1-onevision: Advancing generalized multimodal reasoning through cross-modal formalization","venue":null,"work_id":"b51a0968-8499-4bc8-94f2-c74a46f72b56","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:353187f92a492bcca5cdbd5dac941b48c239d84740ad0edb39354bcbd30e1dfe","observation_id":"ce1f991b-3bb5-4683-84af-b9431f06f8b1","resolution":{"observed_at":"2026-05-13T23:58:29.528009Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.17218","last_updated":"2025-06-20T17:59:31Z","snapshot_observed_at":"2026-07-06T21:45:25.136396Z","submitted_at":"2025-06-20T17:59:31Z","title":"Machine Mental Imagery: Empower Multimodal Reasoning with Latent Visual Tokens","version":1},"cited_work":{"arxiv_id":"2506.17218","doi":"10.48550/arxiv.2506.17218","metadata_source":"pith","pith_arxiv_id":"2506.17218","snapshot_observed_at":"2026-07-10T13:37:06.834211Z","title":"Machine Mental Imagery: Empower Multimodal Reasoning with Latent Visual Tokens","venue":"cs.CV","work_id":"d35f1e96-5f12-4e84-990b-e4b05852180e","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2506.17218","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:4de467ad566e6f9752c02388943cb043e024a848796f66cd987f6ea7b06eb572","observation_id":"570e609f-3dc3-41ae-9f50-edcad274807a","resolution":{"observed_at":"2026-05-13T23:58:28.561653Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.07954","last_updated":"2025-04-10T17:58:27Z","snapshot_observed_at":"2026-07-06T21:07:29.062389Z","submitted_at":"2025-04-10T17:58:27Z","title":"Perception-R1: Pioneering Perception Policy with Reinforcement Learning","version":1},"cited_work":{"arxiv_id":"2504.07954","doi":null,"metadata_source":"pith","pith_arxiv_id":"2504.07954","snapshot_observed_at":"2026-07-05T11:41:02.675605Z","title":"arXiv preprint arXiv:2504.07954 , year =","venue":"cs.CV","work_id":"35656592-ffc7-4aef-9baf-0f8694c0c987","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2504.07954","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:bc2925e26cce52463b2aacd462d2d934625d3c388a6c1c45b5ef194839dc124a","observation_id":"83b88eaf-9ed7-4802-aeb5-e0b6c180f8db","resolution":{"observed_at":"2026-05-13T23:58:28.586668Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Chain-of-focus: Adaptive visual search and zooming for multimodal reasoning via rl.arXiv e-prints, pages arXiv–2505","venue":null,"work_id":"d1f3d4a4-2fb8-4898-a7b2-68d1b6dba1d7","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:7bc5d2f664e3909712fc5f7a0cae4238b5fe925eba587a28f1a248180fa444e3","observation_id":"35f11d4c-8ca6-486e-b4ba-a6a9c0554951","resolution":{"observed_at":"2026-05-13T23:58:29.572747Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.11630","last_updated":"2025-08-15T17:59:49Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-08-15T17:59:49Z","title":"Thyme: Think Beyond Images","version":1},"cited_work":{"arxiv_id":"2508.11630","doi":"10.48550/arxiv.2508.11630","metadata_source":"pith","pith_arxiv_id":"2508.11630","snapshot_observed_at":"2026-07-11T03:17:52.050556Z","title":"Thyme: Think Beyond Images","venue":"cs.CV","work_id":"f91f31cb-6ce5-43a8-b71e-9fc90a2b4160","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2508.11630","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:b622f2a466740ce7c3bbb7494dff98314c4397f2d7eae6953af24187ef2bd046","observation_id":"2f98649b-1879-493f-8174-555efd4cfe0a","resolution":{"observed_at":"2026-05-15T00:33:29.495992Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.13257","last_updated":"2025-02-05T08:44:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-23T17:59:51Z","title":"MME-RealWorld: Could Your Multimodal LLM Challenge High-Resolution Real-World Scenarios that are Difficult for Humans?","version":3},"cited_work":{"arxiv_id":"2408.13257","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.13257","snapshot_observed_at":"2026-07-08T06:34:41.869562Z","title":"MME-RealWorld: Could Your Multimodal LLM Challenge High-Resolution Real-World Scenarios that are Difficult for Humans?","venue":"cs.CV","work_id":"140d79fb-a3d2-4af2-a436-9d997c171f61","year":2024},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2408.13257","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:8169276a436da75f5bb0064f37f0795fc5eea38a8358f12b9ee5a0153380e83a","observation_id":"aeba2d28-612a-4e41-8a8d-ec6c810ed36b","resolution":{"observed_at":"2026-05-16T07:59:32.958879Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.14362","last_updated":"2026-03-01T04:59:56Z","snapshot_observed_at":"2026-08-02T12:23:34.946873Z","submitted_at":"2025-05-20T13:48:11Z","title":"DeepEyes: Incentivizing \"Thinking with Images\" via Reinforcement Learning","version":3},"cited_work":{"arxiv_id":"2505.14362","doi":"10.48550/arxiv.2505.14362","metadata_source":"pith","pith_arxiv_id":"2505.14362","snapshot_observed_at":"2026-07-10T12:15:01.137692Z","title":"DeepEyes: Incentivizing \"Thinking with Images\" via Reinforcement Learning","venue":"cs.CV","work_id":"5f6cf57b-2407-4127-b39c-d8a61494e474","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2505.14362","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:74c94e505f3a421304204ae6e669785e43dfc707c7255b5f7406744c89df7c5d","observation_id":"4695c34f-1a6e-4728-be36-9b5ebaee0ced","resolution":{"observed_at":"2026-05-13T23:58:28.607162Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.10479","last_updated":"2025-04-19T03:47:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-14T17:59:25Z","title":"InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models","version":3},"cited_work":{"arxiv_id":"2504.10479","doi":"10.48550/arxiv.2504.10479","metadata_source":"pith","pith_arxiv_id":"2504.10479","snapshot_observed_at":"2026-07-11T03:17:51.831519Z","title":"InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models","venue":"cs.CV","work_id":"fe8637aa-12bc-4434-8d36-9f57b5eebcbe","year":2025},"citing_paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-13T23:57:47.657243Z"},"links":{"cited_paper":"/paper/2504.10479","citing_paper":"/paper/2604.03307"},"observation_digest":"sha256:ac80af093d205ff08d7c9660e71c455ee148f57b68f4df5920f00a384655c690","observation_id":"e0907d01-1ba7-4334-a8b7-d8447b7d2acf","resolution":{"observed_at":"2026-05-13T23:58:28.497797Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-02T06:30:47.504484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-05-20T07:54:09.017512+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T07:54:09.017512+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2604.03307","last_updated":"2026-04-16T08:04:16Z","latest_version":2,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-31T03:57:56Z","title":"V-Reflection: Transforming MLLMs from Passive Observers to Active Interrogators"},"reference_resolution":{"displayed":40,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":1,"verified_exact":26,"verified_fuzzy":13},"total_outbound_references":40},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-02T06:30:47.504484+00:00","source":"crossref"},{"observed_at":"2026-08-02T06:30:44.765945+00:00","source":"retraction_watch"}],"thesis":"As of 2 August 2026, this Paper Citation Record lists 40 of 40 outbound references and 3 inbound Pith citation observations for arXiv:2604.03307."}