{"as_of":"2026-08-09T06:22:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:4ee905972a61f469ae901a33555f044e1f833242f7a99da059bd5ca9a0504a2b","coverage":[{"denominator":36,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":36,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-26T18:37:27.171215Z","state":"measured"},{"denominator":36,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":36,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2606.19965/citation-record","integrity":"/paper/2606.19965/integrity","json":"/paper/2606.19965/citation-record.json","paper":"/paper/2606.19965"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T18:37:27.171215Z","title":"Scaling Learning Algorithms Towards","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.19965","last_updated":"2026-06-18T09:05:48Z","snapshot_observed_at":"2026-08-01T19:58:30.403562Z","submitted_at":"2026-06-18T09:05:48Z","title":"ROSE: Benchmarking the Perception-to-Action Gap in Multimodal Models","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-06-26T18:37:27.171215Z"},"links":{"citing_paper":"/paper/2606.19965"},"observation_digest":"sha256:0c66f1f79d196df4ab05a713a593fbd1af88f01040cc0f24349aac44847b76dd","observation_id":"00a5e44c-a6f2-4af0-a8dd-0203bca8404f","resolution":{"observed_at":"2026-06-26T18:37:27.171215Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T18:37:27.171215Z","title":"and Osindero, Simon and Teh, Yee Whye , journal =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.19965","last_updated":"2026-06-18T09:05:48Z","snapshot_observed_at":"2026-08-01T19:58:30.403562Z","submitted_at":"2026-06-18T09:05:48Z","title":"ROSE: Benchmarking the Perception-to-Action Gap in Multimodal Models","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-06-26T18:37:27.171215Z"},"links":{"citing_paper":"/paper/2606.19965"},"observation_digest":"sha256:8696d1c0316284460a4f7f91ea71fe164c3ddf1c0188bea6ccd55e04012b7dca","observation_id":"f489f4eb-1908-4b17-bbba-6e2aa937d5d8","resolution":{"observed_at":"2026-06-26T18:37:27.171215Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T18:37:27.171215Z","title":"2016 , publisher=","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2606.19965","last_updated":"2026-06-18T09:05:48Z","snapshot_observed_at":"2026-08-01T19:58:30.403562Z","submitted_at":"2026-06-18T09:05:48Z","title":"ROSE: Benchmarking the Perception-to-Action Gap in Multimodal Models","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-06-26T18:37:27.171215Z"},"links":{"citing_paper":"/paper/2606.19965"},"observation_digest":"sha256:86505a30fa574819a1984fdf4db441788abed142e713ff2d3d73c01e38174d7d","observation_id":"48484566-af6c-421b-93f7-942935b969e0","resolution":{"observed_at":"2026-06-26T18:37:27.171215Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":"2303.08774","doi":"10.1002/tea.20265","metadata_source":"pith","pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GPT-4 Technical Report","venue":"cs.CL","work_id":"b928e041-6991-4c08-8c81-0359e4097c7b","year":2023},"citing_paper":{"arxiv_id":"2606.19965","last_updated":"2026-06-18T09:05:48Z","snapshot_observed_at":"2026-08-01T19:58:30.403562Z","submitted_at":"2026-06-18T09:05:48Z","title":"ROSE: Benchmarking the Perception-to-Action Gap in Multimodal Models","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-06-26T18:37:27.171215Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2606.19965"},"observation_digest":"sha256:f819721041d24b42145142296876a55aeeff8deabb0298a63ac7065d533c6cd7","observation_id":"13ae1fbe-5a4f-42d5-822e-9d67d7944e0d","resolution":{"observed_at":"2026-07-04T02:59:26.142060Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21276","last_updated":"2024-10-25T17:43:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-25T17:43:01Z","title":"GPT-4o System Card","version":1},"cited_work":{"arxiv_id":"2410.21276","doi":"10.1177/15248380231178756","metadata_source":"pith","pith_arxiv_id":"2410.21276","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GPT-4o System Card","venue":"cs.CL","work_id":"f37bf1c7-4964-4e56-9762-d20da8d9009f","year":2024},"citing_paper":{"arxiv_id":"2606.19965","last_updated":"2026-06-18T09:05:48Z","snapshot_observed_at":"2026-08-01T19:58:30.403562Z","submitted_at":"2026-06-18T09:05:48Z","title":"ROSE: Benchmarking the Perception-to-Action Gap in Multimodal Models","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-06-26T18:37:27.171215Z"},"links":{"cited_paper":"/paper/2410.21276","citing_paper":"/paper/2606.19965"},"observation_digest":"sha256:e299d6158bf61501fe133c8d8ad2985a07525f3b617714cc7efb346b716ee101","observation_id":"6cd26660-e967-4366-b90d-212dbb007261","resolution":{"observed_at":"2026-07-04T02:59:26.139173Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":"2312.11805","doi":"10.1038/nrn2888","metadata_source":"pith","pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gemini: A Family of Highly Capable Multimodal Models","venue":"cs.CL","work_id":"83f7c85b-3f11-450f-ac0c-64d9745220b2","year":2023},"citing_paper":{"arxiv_id":"2606.19965","last_updated":"2026-06-18T09:05:48Z","snapshot_observed_at":"2026-08-01T19:58:30.403562Z","submitted_at":"2026-06-18T09:05:48Z","title":"ROSE: Benchmarking the Perception-to-Action Gap in Multimodal Models","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-06-26T18:37:27.171215Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2606.19965"},"observation_digest":"sha256:be1033cec13d0bff2ffd43d58682cd0b4bb884f93a5949f24d1447bcb903d7b5","observation_id":"6aeabeec-95e1-4f19-95ca-297c0d43249c","resolution":{"observed_at":"2026-07-04T02:59:26.148947Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2511.21631","last_updated":"2025-11-27T12:16:54Z","snapshot_observed_at":"2026-07-06T22:37:03.716474Z","submitted_at":"2025-11-26T17:59:08Z","title":"Qwen3-VL Technical Report","version":2},"cited_work":{"arxiv_id":"2511.21631","doi":"10.1016/j.neunet.2025.107777","metadata_source":"pith","pith_arxiv_id":"2511.21631","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen3-VL Technical Report","venue":"cs.CV","work_id":"1fe243aa-e3c0-4da6-b391-4cbcfc88d5c0","year":2025},"citing_paper":{"arxiv_id":"2606.19965","last_updated":"2026-06-18T09:05:48Z","snapshot_observed_at":"2026-08-01T19:58:30.403562Z","submitted_at":"2026-06-18T09:05:48Z","title":"ROSE: Benchmarking the Perception-to-Action Gap in Multimodal Models","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-06-26T18:37:27.171215Z"},"links":{"cited_paper":"/paper/2511.21631","citing_paper":"/paper/2606.19965"},"observation_digest":"sha256:11558f78fe1647a9da435b1ea3db2d4e4e750be8f9cb91e4c87a9c06994a70a3","observation_id":"541f6e75-b5a7-4eca-acfe-a76fdf005bbc","resolution":{"observed_at":"2026-07-04T02:59:26.146079Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.15804","last_updated":"2026-04-21T03:35:14Z","snapshot_observed_at":"2026-08-01T22:56:50.755050Z","submitted_at":"2026-04-17T08:05:46Z","title":"Qwen3.5-Omni Technical Report","version":2},"cited_work":{"arxiv_id":"2604.15804","doi":"10.48550/arxiv.2604.15804","metadata_source":"pith","pith_arxiv_id":"2604.15804","snapshot_observed_at":"2026-08-05T02:49:54.815029Z","title":"Qwen3.5-Omni Technical Report","venue":"cs.CL","work_id":"9d0aeac4-3b94-4a09-af62-5e32286fdf5f","year":2026},"citing_paper":{"arxiv_id":"2606.19965","last_updated":"2026-06-18T09:05:48Z","snapshot_observed_at":"2026-08-01T19:58:30.403562Z","submitted_at":"2026-06-18T09:05:48Z","title":"ROSE: Benchmarking the Perception-to-Action Gap in Multimodal Models","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-06-26T18:37:27.171215Z"},"links":{"cited_paper":"/paper/2604.15804","citing_paper":"/paper/2606.19965"},"observation_digest":"sha256:9acc5b69da300d9c726407542a9df2a95230764121717ae2afc8eca7eff0d245","observation_id":"33edcc4f-f5ed-492c-b6b8-ae8bfcfae151","resolution":{"observed_at":"2026-07-04T02:59:26.115155Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T18:37:27.171215Z","title":"Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR) , month =","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.19965","last_updated":"2026-06-18T09:05:48Z","snapshot_observed_at":"2026-08-01T19:58:30.403562Z","submitted_at":"2026-06-18T09:05:48Z","title":"ROSE: Benchmarking the Perception-to-Action Gap in Multimodal Models","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-06-26T18:37:27.171215Z"},"links":{"citing_paper":"/paper/2606.19965"},"observation_digest":"sha256:61ee515ab57fc7b1f8568ac1cdd004500928d0bcb672a386697730e66f51c2d0","observation_id":"99a99f63-bb8a-4d72-954e-203c5251c624","resolution":{"observed_at":"2026-06-26T18:37:27.171215Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T18:37:27.171215Z","title":"Ref-Adv: Exploring","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2606.19965","last_updated":"2026-06-18T09:05:48Z","snapshot_observed_at":"2026-08-01T19:58:30.403562Z","submitted_at":"2026-06-18T09:05:48Z","title":"ROSE: Benchmarking the Perception-to-Action Gap in Multimodal Models","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-06-26T18:37:27.171215Z"},"links":{"citing_paper":"/paper/2606.19965"},"observation_digest":"sha256:7e624f850a896a365c64d27751af22a2604623487892fc196d2fdf44e64eb31d","observation_id":"37c1dbc0-d586-40b6-80ba-95d3792ed1d1","resolution":{"observed_at":"2026-06-26T18:37:27.171215Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T18:37:27.171215Z","title":"Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR) , month =","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2606.19965","last_updated":"2026-06-18T09:05:48Z","snapshot_observed_at":"2026-08-01T19:58:30.403562Z","submitted_at":"2026-06-18T09:05:48Z","title":"ROSE: Benchmarking the Perception-to-Action Gap in Multimodal Models","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-06-26T18:37:27.171215Z"},"links":{"citing_paper":"/paper/2606.19965"},"observation_digest":"sha256:866f51e1bfd1af71147466983cafd1f4c5f936d21be5e9526f47ec03f8c414f9","observation_id":"459e4827-4e11-4608-9914-519eddf067c4","resolution":{"observed_at":"2026-06-26T18:37:27.171215Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T18:37:27.171215Z","title":"Proceedings of the AAAI Conference on Artificial Intelligence , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.19965","last_updated":"2026-06-18T09:05:48Z","snapshot_observed_at":"2026-08-01T19:58:30.403562Z","submitted_at":"2026-06-18T09:05:48Z","title":"ROSE: Benchmarking the Perception-to-Action Gap in Multimodal Models","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-06-26T18:37:27.171215Z"},"links":{"citing_paper":"/paper/2606.19965"},"observation_digest":"sha256:f39156c6ba93535068ce8814d06705387af00df163658dda6e5f93c1e795aca1","observation_id":"16232f52-2aeb-4a6e-847d-087b8e54fefd","resolution":{"observed_at":"2026-06-26T18:37:27.171215Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T18:37:27.171215Z","title":"Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition , pages=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.19965","last_updated":"2026-06-18T09:05:48Z","snapshot_observed_at":"2026-08-01T19:58:30.403562Z","submitted_at":"2026-06-18T09:05:48Z","title":"ROSE: Benchmarking the Perception-to-Action Gap in Multimodal Models","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-06-26T18:37:27.171215Z"},"links":{"citing_paper":"/paper/2606.19965"},"observation_digest":"sha256:c67a3edfaf44e379acb4a8bf8572faff58724f6729e80d98f27a5e340a1736c3","observation_id":"09d3fe6c-c8e3-424a-a7d6-fe6156626684","resolution":{"observed_at":"2026-06-26T18:37:27.171215Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2602.10138","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T02:59:26.150277Z","title":"arXiv preprint arXiv:2602.10138 , year=","venue":null,"work_id":"8ad0bf37-021a-46d3-ab60-0066e80c4970","year":null},"citing_paper":{"arxiv_id":"2606.19965","last_updated":"2026-06-18T09:05:48Z","snapshot_observed_at":"2026-08-01T19:58:30.403562Z","submitted_at":"2026-06-18T09:05:48Z","title":"ROSE: Benchmarking the Perception-to-Action Gap in Multimodal Models","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-06-26T18:37:27.171215Z"},"links":{"citing_paper":"/paper/2606.19965"},"observation_digest":"sha256:9ad16d4d6349dd32dbce102a55177760eafcc8eec37c05dc3376d1c60d2b0adb","observation_id":"14324db1-861b-4814-b1f6-71ab12fffc2d","resolution":{"observed_at":"2026-07-04T02:59:26.152642Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2511.15090","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T15:09:54.724121Z","title":"Sciegqa: A dataset for scientific evidence-grounded question answering and reasoning","venue":null,"work_id":"3e313ee9-8724-4105-9021-ca933038de2b","year":2025},"citing_paper":{"arxiv_id":"2606.19965","last_updated":"2026-06-18T09:05:48Z","snapshot_observed_at":"2026-08-01T19:58:30.403562Z","submitted_at":"2026-06-18T09:05:48Z","title":"ROSE: Benchmarking the Perception-to-Action Gap in Multimodal Models","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-06-26T18:37:27.171215Z"},"links":{"citing_paper":"/paper/2606.19965"},"observation_digest":"sha256:e37cbc4be6cabe6b5851873b9cb653f80c60bd14817a1990d7fc8a25a0a82f63","observation_id":"beea45e5-9950-4604-85ab-ecef0a1c5104","resolution":{"observed_at":"2026-07-04T02:59:26.119016Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T18:37:27.171215Z","title":"Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition , pages=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.19965","last_updated":"2026-06-18T09:05:48Z","snapshot_observed_at":"2026-08-01T19:58:30.403562Z","submitted_at":"2026-06-18T09:05:48Z","title":"ROSE: Benchmarking the Perception-to-Action Gap in Multimodal Models","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-06-26T18:37:27.171215Z"},"links":{"citing_paper":"/paper/2606.19965"},"observation_digest":"sha256:e79c2fc1ba7834ce6642704d5d33f636e8a61c51c64e7fcca92252efb8fcae16","observation_id":"54dd0633-c3b9-4643-987f-21dab3b0b556","resolution":{"observed_at":"2026-06-26T18:37:27.171215Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T18:37:27.171215Z","title":"The Fourteenth International Conference on Learning Representations , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.19965","last_updated":"2026-06-18T09:05:48Z","snapshot_observed_at":"2026-08-01T19:58:30.403562Z","submitted_at":"2026-06-18T09:05:48Z","title":"ROSE: Benchmarking the Perception-to-Action Gap in Multimodal Models","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-06-26T18:37:27.171215Z"},"links":{"citing_paper":"/paper/2606.19965"},"observation_digest":"sha256:c45555946b1f5aa57be6cae473ac558b7899c5ffcf8a3c917d11e8d9d61758a6","observation_id":"6ef5f344-8e23-48b5-90ab-ccb01d311514","resolution":{"observed_at":"2026-06-26T18:37:27.171215Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T18:37:27.171215Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.19965","last_updated":"2026-06-18T09:05:48Z","snapshot_observed_at":"2026-08-01T19:58:30.403562Z","submitted_at":"2026-06-18T09:05:48Z","title":"ROSE: Benchmarking the Perception-to-Action Gap in Multimodal Models","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-06-26T18:37:27.171215Z"},"links":{"citing_paper":"/paper/2606.19965"},"observation_digest":"sha256:79b91f3a31324c4f06a683a991509739a393e0872c1c3fd50e5b06bf64baef2e","observation_id":"0374ffd3-b362-4d24-a011-013580d8b215","resolution":{"observed_at":"2026-06-26T18:37:27.171215Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2602.12196","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T02:59:26.134392Z","title":"arXiv preprint arXiv:2602.12196 , year=","venue":null,"work_id":"505505d2-66b1-4af5-963a-e9fe9374f93f","year":null},"citing_paper":{"arxiv_id":"2606.19965","last_updated":"2026-06-18T09:05:48Z","snapshot_observed_at":"2026-08-01T19:58:30.403562Z","submitted_at":"2026-06-18T09:05:48Z","title":"ROSE: Benchmarking the Perception-to-Action Gap in Multimodal Models","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-06-26T18:37:27.171215Z"},"links":{"citing_paper":"/paper/2606.19965"},"observation_digest":"sha256:8adfe989abfa85247567e6a5106a67bae94471dd535d7b3944fe206fb7e2cd26","observation_id":"66259507-f9a2-4ee7-97d9-1f253f60a7b3","resolution":{"observed_at":"2026-07-04T02:59:26.136100Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2508.14160","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T02:59:26.130677Z","title":"Rynnec: Bringing mllms into embodied world","venue":null,"work_id":"fb83678f-3928-43d6-baaa-1267aac1e528","year":2025},"citing_paper":{"arxiv_id":"2606.19965","last_updated":"2026-06-18T09:05:48Z","snapshot_observed_at":"2026-08-01T19:58:30.403562Z","submitted_at":"2026-06-18T09:05:48Z","title":"ROSE: Benchmarking the Perception-to-Action Gap in Multimodal Models","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-06-26T18:37:27.171215Z"},"links":{"citing_paper":"/paper/2606.19965"},"observation_digest":"sha256:f11a37e0145da297c02d8a99dbac03aaed0d9dc9ee9682ed48f4c18253b24b84","observation_id":"e0cf176e-6dfe-4f12-9e58-40c373e7ba5c","resolution":{"observed_at":"2026-07-04T02:59:26.132963Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2602.14979","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T11:29:51.440693Z","title":"Rynnbrain: Open embodied foundation models","venue":null,"work_id":"443a7950-489d-487d-a50b-8076ce0dfd48","year":2026},"citing_paper":{"arxiv_id":"2606.19965","last_updated":"2026-06-18T09:05:48Z","snapshot_observed_at":"2026-08-01T19:58:30.403562Z","submitted_at":"2026-06-18T09:05:48Z","title":"ROSE: Benchmarking the Perception-to-Action Gap in Multimodal Models","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-06-26T18:37:27.171215Z"},"links":{"citing_paper":"/paper/2606.19965"},"observation_digest":"sha256:ea7fedfec0b22606fc3dfd8a70265673787705c2e7af02caa60d98d14dfe3831","observation_id":"d7c0fdce-c9be-4260-a415-ab877f2325f3","resolution":{"observed_at":"2026-07-04T02:59:26.125037Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T18:37:27.171215Z","title":"Proceedings of The 8th Conference on Robot Learning , pages =","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.19965","last_updated":"2026-06-18T09:05:48Z","snapshot_observed_at":"2026-08-01T19:58:30.403562Z","submitted_at":"2026-06-18T09:05:48Z","title":"ROSE: Benchmarking the Perception-to-Action Gap in Multimodal Models","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-06-26T18:37:27.171215Z"},"links":{"citing_paper":"/paper/2606.19965"},"observation_digest":"sha256:f8c3a5d4867c89a238feb0864e3a5f3457f61f0922294a9a24f5b572155d2a50","observation_id":"7d0539b8-8054-4c36-ab9f-a08e8015c0b4","resolution":{"observed_at":"2026-06-26T18:37:27.171215Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T18:37:27.171215Z","title":"2026 , url=","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2606.19965","last_updated":"2026-06-18T09:05:48Z","snapshot_observed_at":"2026-08-01T19:58:30.403562Z","submitted_at":"2026-06-18T09:05:48Z","title":"ROSE: Benchmarking the Perception-to-Action Gap in Multimodal Models","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-06-26T18:37:27.171215Z"},"links":{"citing_paper":"/paper/2606.19965"},"observation_digest":"sha256:bcaa15c36cb011a81146cc2e8e470c9173614bbed28fab3134b10f819a9a6334","observation_id":"75c5e932-c306-4d84-b1a6-a3d298fbe0ff","resolution":{"observed_at":"2026-06-26T18:37:27.171215Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T18:37:27.171215Z","title":"The Fourteenth International Conference on Learning Representations , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.19965","last_updated":"2026-06-18T09:05:48Z","snapshot_observed_at":"2026-08-01T19:58:30.403562Z","submitted_at":"2026-06-18T09:05:48Z","title":"ROSE: Benchmarking the Perception-to-Action Gap in Multimodal Models","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-06-26T18:37:27.171215Z"},"links":{"citing_paper":"/paper/2606.19965"},"observation_digest":"sha256:d4d183ae047828263762f6a5aa14848a50df2502b07d8584f5c5d05483690353","observation_id":"d8a293ce-3b10-4b86-b8ae-688252c7c2af","resolution":{"observed_at":"2026-06-26T18:37:27.171215Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T18:37:27.171215Z","title":"The Fourteenth International Conference on Learning Representations , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.19965","last_updated":"2026-06-18T09:05:48Z","snapshot_observed_at":"2026-08-01T19:58:30.403562Z","submitted_at":"2026-06-18T09:05:48Z","title":"ROSE: Benchmarking the Perception-to-Action Gap in Multimodal Models","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-06-26T18:37:27.171215Z"},"links":{"citing_paper":"/paper/2606.19965"},"observation_digest":"sha256:25822d2f574f7d2a763e32f2a4e9287703d2874db8be1b4283f4356048312f43","observation_id":"0995f466-4d24-47ff-af0d-1aaba54f1a19","resolution":{"observed_at":"2026-06-26T18:37:27.171215Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23064","last_updated":"2025-04-02T07:10:05Z","snapshot_observed_at":"2026-08-08T13:40:25.733363Z","submitted_at":"2025-03-29T12:50:38Z","title":"VGRP-Bench: Visual Grid Reasoning Puzzle Benchmark for Large Vision-Language Models","version":2},"cited_work":{"arxiv_id":"2503.23064","doi":"10.48550/arxiv.2503.23064","metadata_source":"arxiv_reference","pith_arxiv_id":"2503.23064","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Vgrp-bench: Visual grid reasoning puzzle benchmark for large vision-language models","venue":"ArXiv.org","work_id":"16313383-b6b5-43f8-a5f3-7b7ed04ffca0","year":2025},"citing_paper":{"arxiv_id":"2606.19965","last_updated":"2026-06-18T09:05:48Z","snapshot_observed_at":"2026-08-01T19:58:30.403562Z","submitted_at":"2026-06-18T09:05:48Z","title":"ROSE: Benchmarking the Perception-to-Action Gap in Multimodal Models","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-06-26T18:37:27.171215Z"},"links":{"cited_paper":"/paper/2503.23064","citing_paper":"/paper/2606.19965"},"observation_digest":"sha256:590abc8e74bfa009da72cb6a22a84e8b1d91df5c779c57e7e7608bf703188240","observation_id":"0114579a-b248-47e7-a0ab-997e05354579","resolution":{"observed_at":"2026-07-04T02:59:26.128499Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T18:37:27.171215Z","title":"Forty-second International Conference on Machine Learning , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.19965","last_updated":"2026-06-18T09:05:48Z","snapshot_observed_at":"2026-08-01T19:58:30.403562Z","submitted_at":"2026-06-18T09:05:48Z","title":"ROSE: Benchmarking the Perception-to-Action Gap in Multimodal Models","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-06-26T18:37:27.171215Z"},"links":{"citing_paper":"/paper/2606.19965"},"observation_digest":"sha256:29c56ff7a6cfd6ea254af29955390c3a2ba3fd4fef3517b600632ff7c935bfb8","observation_id":"078e324d-f897-4ff0-aedc-e057c488368f","resolution":{"observed_at":"2026-06-26T18:37:27.171215Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T18:37:27.171215Z","title":"2026 , url=","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2606.19965","last_updated":"2026-06-18T09:05:48Z","snapshot_observed_at":"2026-08-01T19:58:30.403562Z","submitted_at":"2026-06-18T09:05:48Z","title":"ROSE: Benchmarking the Perception-to-Action Gap in Multimodal Models","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-06-26T18:37:27.171215Z"},"links":{"citing_paper":"/paper/2606.19965"},"observation_digest":"sha256:91a728e43efabfba81c15873af858a4df1109bb9729c903f7d7bf3e14b3223f4","observation_id":"54523f20-8be8-4fc0-aa50-f19dc2badf5a","resolution":{"observed_at":"2026-06-26T18:37:27.171215Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T18:37:27.171215Z","title":"2024 , editor =","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.19965","last_updated":"2026-06-18T09:05:48Z","snapshot_observed_at":"2026-08-01T19:58:30.403562Z","submitted_at":"2026-06-18T09:05:48Z","title":"ROSE: Benchmarking the Perception-to-Action Gap in Multimodal Models","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-06-26T18:37:27.171215Z"},"links":{"citing_paper":"/paper/2606.19965"},"observation_digest":"sha256:4977ab6018d9c65445af83de79fe87ee1916ed45bac3cdbd92e7774a98bb9012","observation_id":"45673753-fe77-4d0f-aea5-2a54474f5eed","resolution":{"observed_at":"2026-06-26T18:37:27.171215Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T18:37:27.171215Z","title":"Proceedings of CVPR , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.19965","last_updated":"2026-06-18T09:05:48Z","snapshot_observed_at":"2026-08-01T19:58:30.403562Z","submitted_at":"2026-06-18T09:05:48Z","title":"ROSE: Benchmarking the Perception-to-Action Gap in Multimodal Models","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-06-26T18:37:27.171215Z"},"links":{"citing_paper":"/paper/2606.19965"},"observation_digest":"sha256:fe453d9ebd7e36acb66642031d9cbf9950065159ec8a47cc113b491dc7aaddbd","observation_id":"c4e0ade0-0914-414c-b049-aab12f81d487","resolution":{"observed_at":"2026-06-26T18:37:27.171215Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T18:37:27.171215Z","title":"and Ma, Wei-Chiu and Krishna, Ranjay","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.19965","last_updated":"2026-06-18T09:05:48Z","snapshot_observed_at":"2026-08-01T19:58:30.403562Z","submitted_at":"2026-06-18T09:05:48Z","title":"ROSE: Benchmarking the Perception-to-Action Gap in Multimodal Models","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-06-26T18:37:27.171215Z"},"links":{"citing_paper":"/paper/2606.19965"},"observation_digest":"sha256:b5c0b841abbea9114e27a18d6854a46adbbb6df9ca989414275c334eb1b6a429","observation_id":"e8b01a17-4b02-4434-9b64-4d0fa48f0a47","resolution":{"observed_at":"2026-06-26T18:37:27.171215Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T18:37:27.171215Z","title":"Proceedings of the IEEE/CVF International Conference on Computer Vision (ICCV) , month =","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.19965","last_updated":"2026-06-18T09:05:48Z","snapshot_observed_at":"2026-08-01T19:58:30.403562Z","submitted_at":"2026-06-18T09:05:48Z","title":"ROSE: Benchmarking the Perception-to-Action Gap in Multimodal Models","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-06-26T18:37:27.171215Z"},"links":{"citing_paper":"/paper/2606.19965"},"observation_digest":"sha256:f9cabdf7dba29cbe5748070fe0651d22ed7fbb7a0cc2f452f3e818aed424da7e","observation_id":"a7ba8f5d-f34e-48a8-aa32-eb1ca9be5226","resolution":{"observed_at":"2026-06-26T18:37:27.171215Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T18:37:27.171215Z","title":"Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR) , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.19965","last_updated":"2026-06-18T09:05:48Z","snapshot_observed_at":"2026-08-01T19:58:30.403562Z","submitted_at":"2026-06-18T09:05:48Z","title":"ROSE: Benchmarking the Perception-to-Action Gap in Multimodal Models","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-06-26T18:37:27.171215Z"},"links":{"citing_paper":"/paper/2606.19965"},"observation_digest":"sha256:c06cbb5ff6388307950290f935035e2d28f916b1c7c71a9ea58c6644a7835b87","observation_id":"695de61f-fee8-4c04-b671-77943ec5e307","resolution":{"observed_at":"2026-06-26T18:37:27.171215Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.13311","last_updated":"2024-07-16T03:36:29Z","snapshot_observed_at":"2026-07-06T17:19:48.251975Z","submitted_at":"2024-01-24T09:07:11Z","title":"ConTextual: Evaluating Context-Sensitive Text-Rich Visual Reasoning in Large Multimodal Models","version":3},"cited_work":{"arxiv_id":"2401.13311","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.13311","snapshot_observed_at":"2026-07-04T02:59:26.120279Z","title":"Contextual: Evaluating context- sensitive text-rich visual reasoning in large multimodal models","venue":null,"work_id":"3680e5a2-aa1a-40cd-936e-77e6ef0b7686","year":2024},"citing_paper":{"arxiv_id":"2606.19965","last_updated":"2026-06-18T09:05:48Z","snapshot_observed_at":"2026-08-01T19:58:30.403562Z","submitted_at":"2026-06-18T09:05:48Z","title":"ROSE: Benchmarking the Perception-to-Action Gap in Multimodal Models","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-06-26T18:37:27.171215Z"},"links":{"cited_paper":"/paper/2401.13311","citing_paper":"/paper/2606.19965"},"observation_digest":"sha256:e499d030a8138b4edad04af25a8bad884fb4b2b792acaf1835f72fb104f88c50","observation_id":"cb42289b-853e-4eec-b5b2-cf6c36860f02","resolution":{"observed_at":"2026-07-04T02:59:26.121974Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2024.acl-long.573","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CODIS : Benchmarking Context-dependent Visual Comprehension for Multimodal Large Language Models","venue":null,"work_id":"634ccc18-8950-4baf-8731-d355a4707d8b","year":2024},"citing_paper":{"arxiv_id":"2606.19965","last_updated":"2026-06-18T09:05:48Z","snapshot_observed_at":"2026-08-01T19:58:30.403562Z","submitted_at":"2026-06-18T09:05:48Z","title":"ROSE: Benchmarking the Perception-to-Action Gap in Multimodal Models","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-06-26T18:37:27.171215Z"},"links":{"citing_paper":"/paper/2606.19965"},"observation_digest":"sha256:2a795253746bafac538112464c16327b763ccc6027bb75f08d84b180258e05ca","observation_id":"cef3d664-722f-4214-8ba4-3cdc91fde2cd","resolution":{"observed_at":"2026-06-26T18:39:43.074903Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-26T18:37:27.171215Z","title":"International Conference on Learning Representations , volume=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.19965","last_updated":"2026-06-18T09:05:48Z","snapshot_observed_at":"2026-08-01T19:58:30.403562Z","submitted_at":"2026-06-18T09:05:48Z","title":"ROSE: Benchmarking the Perception-to-Action Gap in Multimodal Models","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-06-26T18:37:27.171215Z"},"links":{"citing_paper":"/paper/2606.19965"},"observation_digest":"sha256:b754b4fff3309a94b8583a480254e6392aefc595eae7427a77edbedf1ab46eff","observation_id":"82d1fd7e-1174-469c-88af-c7b2dcc62d11","resolution":{"observed_at":"2026-06-26T18:37:27.171215Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2606.19965","last_updated":"2026-06-18T09:05:48Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-01T19:58:30.403562Z","submitted_at":"2026-06-18T09:05:48Z","title":"ROSE: Benchmarking the Perception-to-Action Gap in Multimodal Models"},"reference_resolution":{"displayed":36,"state_counts":{"malformed_identifier":0,"metadata_mismatch":10,"parse_uncertain":0,"unresolved":23,"verified_exact":3,"verified_fuzzy":0},"total_outbound_references":36},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 36 of 36 outbound references and 0 inbound Pith citation observations for arXiv:2606.19965."}