{"as_of":"2026-08-07T18:00:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:a97272f368a9b9a685b24a9edc314003b252e5998b994e35f4bd8c085762104d","coverage":[{"denominator":52,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":52,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-28T02:25:30.998989Z","state":"measured"},{"denominator":54,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":54,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T04:18:02.360277Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-06T00:31:39.011940Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"cited_work":{"arxiv_id":"2606.06476","doi":null,"metadata_source":"pith","pith_arxiv_id":"2606.06476","snapshot_observed_at":"2026-08-06T00:31:39.011940Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","venue":"cs.CV","work_id":"9366708e-3ccb-495a-9209-3af28837e6c2","year":2026},"citing_paper":{"arxiv_id":"2608.01207","last_updated":"2026-08-05T05:56:24Z","snapshot_observed_at":"2026-08-07T17:49:08.580112Z","submitted_at":"2026-08-02T12:47:39Z","title":"It's the Decoding Format, Not the Perturbation: Auditing Consistency-Based Selection for Vision-Language Test-Time Scaling","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T00:31:38.919997Z"},"links":{"cited_paper":"/paper/2606.06476","citing_paper":"/paper/2608.01207"},"observation_digest":"sha256:af49f8bd3c144ca037669c84ea198f6e4693961a8e2190271356687efe37070a","observation_id":"b34d7c2c-182d-4197-b6a5-c2eee8a6e692","resolution":{"observed_at":"2026-08-06T00:31:39.103493Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2606.06476","snapshot_observed_at":"2026-08-06T04:18:02.360277Z","title":"Thinking with imagination: Agentic visual spatial reasoning with world simulators","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.01207","last_updated":"2026-08-05T05:56:24Z","snapshot_observed_at":"2026-08-07T17:49:08.580112Z","submitted_at":"2026-08-02T12:47:39Z","title":"It's the Decoding Format, Not the Perturbation: Auditing Consistency-Based Selection for Vision-Language Test-Time Scaling","version":2},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-06T04:18:02.360277Z"},"links":{"cited_paper":"/paper/2606.06476","citing_paper":"/paper/2608.01207"},"observation_digest":"sha256:c84a74897ce9b6d5372280645d758713c333e6b891ca4273a90b9d22f3d4dfa1","observation_id":"a2c27826-25f7-4ef7-a988-846aa444833e","resolution":{"observed_at":"2026-08-06T04:18:02.360277Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2606.06476/citation-record","integrity":"/paper/2606.06476/integrity","json":"/paper/2606.06476/citation-record.json","paper":"/paper/2606.06476"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2506.21458","doi":"10.48550/arxiv.2506.21458","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Spa- tial mental modeling from limited views","venue":"arXiv (Cornell University)","work_id":"e7da71d2-40fb-4670-9c50-b37e7469d519","year":2025},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:4f8eca09bcbd09ae7725f8c6eddd023f11b5e73b5aa4ecb1a619bb156acc8141","observation_id":"5a569d20-733a-4ae9-b718-895c9b9881ef","resolution":{"observed_at":"2026-07-02T12:06:56.229889Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T02:25:30.998989Z","title":"Mmsi-bench: A benchmark for multi-image spatial intelligence","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:1453a5c5ca9ecaaba8f5c5adf636d84eefb3f02b98cc7f853d197c4eeb5990c7","observation_id":"8dc8ff68-4e84-4717-8037-fc2cd61fefc3","resolution":{"observed_at":"2026-06-28T02:25:30.998989Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2507.07984","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T01:27:30.587936Z","title":"Ost-bench: Evaluating the capabilities of mllms in online spatio-temporal scene understanding","venue":null,"work_id":"ca782bad-f2ab-48af-9cca-a30e5aebf8e2","year":2025},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:527575ff549ad5b9184db2c19afc74a92bc2e827e2c9e03200d5c551de8c17e2","observation_id":"a854cf28-5076-4bec-8c69-cfed6ac971e0","resolution":{"observed_at":"2026-07-02T12:06:56.232614Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2512.10863","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T10:48:03.213683Z","title":"Mmsi-video-bench: A holistic benchmark for video-based spatial intelligence","venue":null,"work_id":"5576e252-6316-45fb-a511-bcbf65609546","year":2025},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:c29af9b710d295b61c242e9a17ab194affa171aafbba78cc483d06e822c51718","observation_id":"508d67f2-2e5e-47e6-b417-43c33b3cb318","resolution":{"observed_at":"2026-07-02T12:06:56.193731Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2505.21500","doi":"10.48550/arxiv.2505.21500","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Viewspatial-bench: Evaluating multi-perspective spatial localization in vision-language models.ArXiv, abs/2505.21500","venue":"ArXiv.org","work_id":"4e64344e-47c6-4d04-a5a4-c05266da7d8c","year":2025},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:999f28e12d2516d28b614b073e483f506c541031083069f413faaa37494f0579","observation_id":"4bc0e1e7-5eb0-472d-b00f-8e800bd712bc","resolution":{"observed_at":"2026-07-02T12:06:56.195938Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2412.07825","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T03:29:31.552020Z","title":"3dsrbench: A comprehensive 3d spatial reasoning benchmark","venue":null,"work_id":"3d100652-7ff3-4c05-a9cb-9f7383953f74","year":2024},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:78a20ee151d0e875713aab9dc62c876d45b23917a9fc6b7860152bab0634e4b8","observation_id":"0db002b8-22ef-45c7-bc18-bd9fadd47c66","resolution":{"observed_at":"2026-07-02T12:06:56.191022Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.14171","last_updated":"2025-07-02T21:00:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-18T18:59:54Z","title":"Thinking in Space: How Multimodal Large Language Models See, Remember, and Recall Spaces","version":2},"cited_work":{"arxiv_id":"2412.14171","doi":null,"metadata_source":"pith","pith_arxiv_id":"2412.14171","snapshot_observed_at":"2026-07-04T06:29:37.915954Z","title":"Thinking in Space: How Multimodal Large Language Models See, Remember, and Recall Spaces","venue":"cs.CV","work_id":"bbaf0f51-258a-4d79-b5a4-5b1e9120b2c0","year":2024},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"cited_paper":"/paper/2412.14171","citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:b2c5dd19e0990f71acfc6b97aa549070291416d12640a3c4e96db02062374928","observation_id":"920f5343-a720-43dd-869f-be89087e9f5a","resolution":{"observed_at":"2026-07-02T12:06:56.204306Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2512.24330","doi":"10.48550/arxiv.2512.24330","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sensenova-mars: Empowering multimodal agentic reasoning and search via reinforcement learning","venue":"arXiv (Cornell University)","work_id":"f5b58e73-00a7-404d-8d5d-d72dd8b11531","year":2025},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:fc7b266ef7441266ae8ecb8f88c0e6ff11a8a6703786e0fa0fc34d16ff44426a","observation_id":"7831847e-d629-4a0f-b207-5f6b614c08a6","resolution":{"observed_at":"2026-07-02T12:06:56.164844Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2511.05491","doi":"10.48550/arxiv.2511.05491","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Thinking in space: How multimodal large language models see, remember, and recall spaces","venue":"arXiv (Cornell University)","work_id":"98823083-5573-413f-a371-f50911725458","year":2025},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:42bc0ddd4c819b45219939ae49fea09b4b92429587e738799b22c8bfa81563f7","observation_id":"fd55e28a-fd42-4967-8011-cb46d319c450","resolution":{"observed_at":"2026-07-02T12:06:56.193385Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.18125","last_updated":"2025-04-27T06:50:23Z","snapshot_observed_at":"2026-07-06T19:22:58.341345Z","submitted_at":"2024-09-26T17:59:11Z","title":"LLaVA-3D: A Simple yet Effective Pathway to Empowering LMMs with 3D-awareness","version":3},"cited_work":{"arxiv_id":"2409.18125","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2409.18125","snapshot_observed_at":"2026-07-04T15:39:56.857028Z","title":"Llava-3d: A simple yet effective pathway to empowering lmms with 3d-awareness","venue":null,"work_id":"25557a62-345c-4dfe-a582-4ff77ad59e6a","year":2024},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"cited_paper":"/paper/2409.18125","citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:0b828f92d2d34c79d65f6047abf0473508cfb66066719bf8751a2db892b44bd5","observation_id":"a27de77b-f796-415c-8585-73b8247538b8","resolution":{"observed_at":"2026-07-02T12:06:56.162004Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.01901","last_updated":"2025-04-02T16:59:55Z","snapshot_observed_at":"2026-08-07T16:14:03.001458Z","submitted_at":"2025-04-02T16:59:55Z","title":"Ross3D: Reconstructive Visual Instruction Tuning with 3D-Awareness","version":1},"cited_work":{"arxiv_id":"2504.01901","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2504.01901","snapshot_observed_at":"2026-07-04T16:39:57.809937Z","title":"Ross3d: Recon- structive visual instruction tuning with 3d-awareness","venue":null,"work_id":"d3ba3e21-3bf6-414c-840c-1d39858fe7f0","year":2025},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"cited_paper":"/paper/2504.01901","citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:a238cfd953af4576bea9ac1ecd18228363f4b7d46d87acdce925426dc88d97c9","observation_id":"69f14586-43f4-4ac8-8045-a4d963e96b63","resolution":{"observed_at":"2026-07-02T12:06:56.220175Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T02:25:30.998989Z","title":"Vlm-3r: Vision-language models augmented with instruction-aligned 3d reconstruction,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:421c0159101a1f688a19a886a07588154a2762cea435db61249943948a223988","observation_id":"50f47dc9-3677-4e1e-b2e7-cf8488863fc8","resolution":{"observed_at":"2026-06-28T02:25:30.998989Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.20279","last_updated":"2026-04-21T02:48:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-26T17:56:30Z","title":"VLM-3R: Vision-Language Models Augmented with Instruction-Aligned 3D Reconstruction","version":5},"cited_work":{"arxiv_id":"2505.20279","doi":"10.48550/arxiv.2505.20279","metadata_source":"pith","pith_arxiv_id":"2505.20279","snapshot_observed_at":"2026-08-05T02:49:54.815029Z","title":"VLM-3R: Vision-Language Models Augmented with Instruction-Aligned 3D Reconstruction","venue":"cs.CV","work_id":"1b2c1dfb-9d76-4c59-bcfa-8f23e957c138","year":2025},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"cited_paper":"/paper/2505.20279","citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:d6ee2ece71f8e8aaa616cbf391efdcc62842108cff055e9aec6f021a61cbeb2e","observation_id":"2ad5078d-463c-4a61-8820-64f322518b5e","resolution":{"observed_at":"2026-07-02T12:06:56.185660Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2511.21688","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T06:07:41.733757Z","title":"g2vlm: Geometry grounded vision language model with unified 3d reconstruction and spatial reasoning.arXiv preprint arXiv:2511.21688","venue":null,"work_id":"d9da8e99-dfd3-4391-b8ae-f66459ed28b6","year":2025},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:caf632cb1e2c3ca6ec599e3cd89d01a68d3e69c775beaf2ff506a6e6cb6cfeef","observation_id":"2e6db3c7-5807-4a80-a236-6f90a889be73","resolution":{"observed_at":"2026-07-02T12:06:56.227422Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2511.22659","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T08:59:43.203750Z","title":"Geometrically-constrained agent for spatial reasoning","venue":null,"work_id":"4af12e4e-d3e7-47d4-b309-1347a9867877","year":2024},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:515afd508bfb17ebb85a2c8fb8d44aefb00061fdfb68682c0f42ec447d00b107","observation_id":"fbccb9d8-640c-4a81-9ec4-eb2ba89d4745","resolution":{"observed_at":"2026-07-02T12:06:56.181153Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.07181","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T12:04:50.506641Z","title":"Tiger: Tool-integrated geometric rea- soning in vision-language models for robotics","venue":null,"work_id":"07e7b1f7-5efa-40e2-94ed-12b94b8e73cf","year":2025},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:3c8a70f274a82b8f96ea8653b99f9cc3c6d1dc70e17a4d49666fe4f31d3ce2b1","observation_id":"4f050af7-b1df-4013-b74e-19a13fad355e","resolution":{"observed_at":"2026-07-02T12:06:56.215379Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2512.04563","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-02T12:06:56.199633Z","title":"Cooper: A unified model for cooperative perception and reasoning in spatial intelligence","venue":null,"work_id":"0d8297a1-8164-4e0d-9faf-937e1bcc4b89","year":2025},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:a2196de76679c10f65905ea8d00ea4b45ed684a1374f5dc6ead05f6532459911","observation_id":"35ab1a64-957d-4f9d-95c4-feb01e6632b7","resolution":{"observed_at":"2026-07-02T12:06:56.201101Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T02:25:30.998989Z","title":"Introducing o3 and o4 mini.https://openai.com/index/introducing-o3-and-o4-mini/, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:8afdfd03b2c063d491da9f9a40986f1116192fbd8b97f2d39b35ed81d817aaed","observation_id":"8db70bc4-f904-4b89-9e6f-763f9e1843e9","resolution":{"observed_at":"2026-06-28T02:25:30.998989Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2511.05271","last_updated":"2026-03-11T08:46:41Z","snapshot_observed_at":"2026-07-06T22:35:13.699594Z","submitted_at":"2025-11-07T14:31:20Z","title":"DeepEyesV2: Toward Agentic Multimodal Model","version":4},"cited_work":{"arxiv_id":"2511.05271","doi":"10.48550/arxiv.2511.05271","metadata_source":"pith","pith_arxiv_id":"2511.05271","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DeepEyesV2: Toward Agentic Multimodal Model","venue":"cs.CV","work_id":"0bc8f779-a287-4c6f-95b2-daffd15ad044","year":2025},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"cited_paper":"/paper/2511.05271","citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:89cc6cdbcfa6f51fb9deec30da1873842d39230161b4ded8fa1de95fffce51f8","observation_id":"03782742-5179-4a89-ae9b-b613271964ec","resolution":{"observed_at":"2026-07-02T12:06:56.222490Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.23918","last_updated":"2025-07-03T16:49:39Z","snapshot_observed_at":"2026-07-06T21:49:47.325553Z","submitted_at":"2025-06-30T14:48:35Z","title":"Thinking with Images for Multimodal Reasoning: Foundations, Methods, and Future Frontiers","version":3},"cited_work":{"arxiv_id":"2506.23918","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.23918","snapshot_observed_at":"2026-07-11T03:17:51.399835Z","title":"Thinking with Images for Multimodal Reasoning: Foundations, Methods, and Future Frontiers","venue":"cs.CV","work_id":"760ebd7d-977d-4280-afae-adb421d49ed4","year":2025},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"cited_paper":"/paper/2506.23918","citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:dcb29aa4a775ce63213e5bf4185b7c892097509326ae4fa7637848476f20035d","observation_id":"b3da04b2-3d44-40f7-9d55-3622574e849e","resolution":{"observed_at":"2026-07-02T12:06:56.167205Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.11630","last_updated":"2025-08-15T17:59:49Z","snapshot_observed_at":"2026-08-03T04:02:35.392835Z","submitted_at":"2025-08-15T17:59:49Z","title":"Thyme: Think Beyond Images","version":1},"cited_work":{"arxiv_id":"2508.11630","doi":"10.48550/arxiv.2508.11630","metadata_source":"pith","pith_arxiv_id":"2508.11630","snapshot_observed_at":"2026-07-11T03:17:52.050556Z","title":"Thyme: Think Beyond Images","venue":"cs.CV","work_id":"f91f31cb-6ce5-43a8-b71e-9fc90a2b4160","year":2025},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"cited_paper":"/paper/2508.11630","citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:7cb2c6b85bd6d00cd1342a72fa6a4a60599f474e1b586129c6cba4afeb30116d","observation_id":"8bacb188-3a19-4fa2-8a85-5652376ef891","resolution":{"observed_at":"2026-07-02T12:06:56.225105Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.08617","last_updated":"2025-07-09T14:53:06Z","snapshot_observed_at":"2026-07-06T21:23:19.117703Z","submitted_at":"2025-05-13T14:35:51Z","title":"OpenThinkIMG: Learning to Think with Images via Visual Tool Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2505.08617","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.08617","snapshot_observed_at":"2026-07-04T20:10:07.295499Z","title":"OpenThinkIMG: Learning to Think with Images via Visual Tool Reinforcement Learning","venue":"cs.CV","work_id":"3939edf9-d5d5-4c79-bd9c-02e7961fda21","year":2025},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"cited_paper":"/paper/2505.08617","citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:29577fedfb13064adefbfdebcd7118c5f8783cbcd0958422702e8ade5b1f8754","observation_id":"fdcc8cf3-0558-416f-81ad-7e961f6faa53","resolution":{"observed_at":"2026-07-02T12:06:56.198559Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.14362","last_updated":"2026-03-01T04:59:56Z","snapshot_observed_at":"2026-08-02T12:23:34.946873Z","submitted_at":"2025-05-20T13:48:11Z","title":"DeepEyes: Incentivizing \"Thinking with Images\" via Reinforcement Learning","version":3},"cited_work":{"arxiv_id":"2505.14362","doi":"10.48550/arxiv.2505.14362","metadata_source":"pith","pith_arxiv_id":"2505.14362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DeepEyes: Incentivizing \"Thinking with Images\" via Reinforcement Learning","venue":"cs.CV","work_id":"5f6cf57b-2407-4127-b39c-d8a61494e474","year":2025},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"cited_paper":"/paper/2505.14362","citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:45dfab9dbcb76fc776e1519c17c0711a9d9ec0cfa2528c9c04da8eece937d830","observation_id":"3d207530-6e01-4eec-a145-aff8bc33564d","resolution":{"observed_at":"2026-07-02T12:06:56.203349Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.15966","last_updated":"2025-10-24T09:35:22Z","snapshot_observed_at":"2026-07-06T21:28:06.120576Z","submitted_at":"2025-05-21T19:35:08Z","title":"Pixel Reasoner: Incentivizing Pixel-Space Reasoning with Curiosity-Driven Reinforcement Learning","version":3},"cited_work":{"arxiv_id":"2505.15966","doi":"10.48550/arxiv.2505.15966","metadata_source":"pith","pith_arxiv_id":"2505.15966","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Pixel Reasoner: Incentivizing Pixel-Space Reasoning with Curiosity-Driven Reinforcement Learning","venue":"cs.CV","work_id":"878c3e90-ce55-4ba3-a588-2abe369013e6","year":2025},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"cited_paper":"/paper/2505.15966","citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:4d8aecbd04c5558f04b978d03a7ea13019e61f50c59ad3e15a6b61a224985544","observation_id":"86942cc5-d9aa-48e1-b70a-5dea77288517","resolution":{"observed_at":"2026-07-02T12:06:56.214696Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.07969","last_updated":"2025-09-09T17:54:21Z","snapshot_observed_at":"2026-07-30T05:21:39.737665Z","submitted_at":"2025-09-09T17:54:21Z","title":"Mini-o3: Scaling Up Reasoning Patterns and Interaction Turns for Visual Search","version":1},"cited_work":{"arxiv_id":"2509.07969","doi":"10.48550/arxiv.2509.07969","metadata_source":"pith","pith_arxiv_id":"2509.07969","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mini-o3: Scaling Up Reasoning Patterns and Interaction Turns for Visual Search","venue":"cs.CV","work_id":"8d8763c6-6201-4243-9dab-975ba02a78db","year":2025},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"cited_paper":"/paper/2509.07969","citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:418d049822c83658f08c77b42037e55bfa1aa59768262c18c3d2a1f5a20f1a1c","observation_id":"558e4edc-37f8-4125-ad21-07b507ea1b10","resolution":{"observed_at":"2026-07-02T12:06:56.227382Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2601.19834","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T13:38:19.365035Z","title":"Visual generation unlocks human-like reasoning through multimodal world models","venue":null,"work_id":"c93eb550-1241-4a65-ad81-6d0adbd24153","year":2026},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:2581df2a3ade1b3df4dffa6bc6a7d07c2ee33f21e87fe3ef577e85a3bb59239f","observation_id":"9e10e64f-823d-45bd-86de-70c984581e10","resolution":{"observed_at":"2026-07-02T12:06:56.237177Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T02:25:30.998989Z","title":"Isaac sim: Robotics simulation and synthetic data generation.https://developer.nvidia.com/isaac/sim, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:89829b3bda06ef93eece5443a2c865d723c55c9c306986baafea3e72f82f74d5","observation_id":"2f978e7b-6b3f-477e-8cfa-396b58615dd2","resolution":{"observed_at":"2026-06-28T02:25:30.998989Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T02:25:30.998989Z","title":"Scannet++: A high-fidelity dataset of 3d indoor scenes","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:9dbf15ff2379ff39fa06df8140d88156333ebdc7e8431410b37a8f0e1756e674","observation_id":"9a708909-e01c-48b6-8214-c300a482d63f","resolution":{"observed_at":"2026-06-28T02:25:30.998989Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T02:25:30.998989Z","title":"Scannet: Richly-annotated 3d reconstructions of indoor scenes","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:ce6b8450396a97faaa28675517a41950ea037232c00af9963e35adcc01bddd26","observation_id":"7db37277-170a-4ecf-97f1-7f7d926fbc21","resolution":{"observed_at":"2026-06-28T02:25:30.998989Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1709.06158","last_updated":"2017-09-18T20:34:48Z","snapshot_observed_at":"2026-08-02T16:47:23.623541Z","submitted_at":"2017-09-18T20:34:48Z","title":"Matterport3D: Learning from RGB-D Data in Indoor Environments","version":1},"cited_work":{"arxiv_id":"1709.06158","doi":"10.48550/arxiv.1709.06158","metadata_source":"pith","pith_arxiv_id":"1709.06158","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Matterport3D: Learning from RGB-D Data in Indoor Environments","venue":"cs.CV","work_id":"a6675134-1bd7-4d3f-9344-d7072e7449e9","year":2017},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"cited_paper":"/paper/1709.06158","citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:00c08e3e42b391b338ff4f2b5c8ad69f17ac6e96068a5ba3105f59af4b189e30","observation_id":"39573bb8-ff16-4a69-8e08-97f93dcd7e61","resolution":{"observed_at":"2026-07-02T12:06:56.183393Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T02:25:30.998989Z","title":"Dl3dv-10k: A large-scale scene dataset for deep learning-based 3d vision","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:eaa73041c3a337f23e226612ab1413172d9bd781e6a82388f63616450943a1ae","observation_id":"74bc05d2-0c9b-4b7e-9981-09b263cc8ab0","resolution":{"observed_at":"2026-06-28T02:25:30.998989Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T02:25:30.998989Z","title":"ARKitscenes - a diverse real-world dataset for 3d indoor scene understanding using mobile RGB-d data","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:135724ee88b36626c13b7c9ff1abfb098ccb778957f65d280909b4a137b6a50a","observation_id":"bf932a21-e03b-469c-9f51-171f79cd4a6f","resolution":{"observed_at":"2026-06-28T02:25:30.998989Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"9031.369607","doi":"10.1145/3689031.3696072","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Hybridflow: A flexible and efficient rlhf framework","venue":null,"work_id":"4909736c-3fe1-4820-b489-cca51669c6d2","year":2025},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:d4656eaf75afcd6657a652fce5fcf864cda75dff3e1c00e517e524ad4802a3cb","observation_id":"e591fec8-e635-4fd8-be3c-7add44d41e07","resolution":{"observed_at":"2026-06-28T02:31:30.184702Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2602.02204","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T19:40:07.015791Z","title":"vllm-omni: Fully disaggregated serving for any-to-any multimodal models","venue":null,"work_id":"11eb1840-40c6-4f3c-8f6a-53422545f9d8","year":2026},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:4c1e9c440f675e5c558b88fb498561b6b8c50eb7873eabd3f5b1064316678f07","observation_id":"897d13f8-ef9a-48a9-942c-61c294f2aa06","resolution":{"observed_at":"2026-07-02T12:06:56.212793Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2511.21631","last_updated":"2025-11-27T12:16:54Z","snapshot_observed_at":"2026-07-06T22:37:03.716474Z","submitted_at":"2025-11-26T17:59:08Z","title":"Qwen3-VL Technical Report","version":2},"cited_work":{"arxiv_id":"2511.21631","doi":"10.1016/j.neunet.2025.107777","metadata_source":"pith","pith_arxiv_id":"2511.21631","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen3-VL Technical Report","venue":"cs.CV","work_id":"1fe243aa-e3c0-4da6-b391-4cbcfc88d5c0","year":2025},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"cited_paper":"/paper/2511.21631","citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:4aa19a32c6430acebdb45802b9f9e9b909c2ccfa0be105aefe80d7cf61a0fc54","observation_id":"34a83ded-d60d-415f-b209-7bd8fb4c79e7","resolution":{"observed_at":"2026-07-02T12:06:56.234599Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.14683","last_updated":"2025-07-27T11:45:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-20T17:59:30Z","title":"Emerging Properties in Unified Multimodal Pretraining","version":3},"cited_work":{"arxiv_id":"2505.14683","doi":"10.48550/arxiv.2505.14683","metadata_source":"pith","pith_arxiv_id":"2505.14683","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Emerging Properties in Unified Multimodal Pretraining","venue":"cs.CV","work_id":"e0cfd82c-f5d4-44fd-b531-ec73ab0a805b","year":2025},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"cited_paper":"/paper/2505.14683","citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:19cb887792310a351576c3f5d82f9f0484c1ea8c0116933243e3ce53b11cdfaa","observation_id":"5f7c7154-3dc0-4447-9863-f53f7bc7c0d5","resolution":{"observed_at":"2026-07-02T12:06:56.232521Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T02:25:30.998989Z","title":"Spatialllm: A compound 3d-informed design towards spatially-intelligent large multimodal models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:a9373bb0c90f2b76a48a5f078c3d7b9fd447ea7347e18ad51b6eb57605e82436","observation_id":"6684cb35-14f4-4241-a5e6-7347c78139f9","resolution":{"observed_at":"2026-06-28T02:25:30.998989Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.23747","last_updated":"2026-05-19T02:23:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-29T17:59:04Z","title":"Spatial-MLLM: Boosting MLLM Capabilities in Visual-based Spatial Intelligence","version":2},"cited_work":{"arxiv_id":"2505.23747","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.23747","snapshot_observed_at":"2026-07-04T16:39:57.786059Z","title":"Spatial-MLLM: Boosting MLLM Capabilities in Visual-based Spatial Intelligence","venue":"cs.CV","work_id":"389bb6a7-1369-4d6f-a808-838fa4f5b635","year":2025},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"cited_paper":"/paper/2505.23747","citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:9e21c808d8068693861a705d4ae6c521b6c264063774f62e5583fe5ba81506d2","observation_id":"69e8efbf-ddca-4eb3-a01a-a073eee095de","resolution":{"observed_at":"2026-07-02T12:06:56.188190Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.08531","doi":"10.48550/arxiv.2510.08531","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Llava-st: A multimodal large language model for fine-grained spatial-temporal understanding","venue":"ArXiv.org","work_id":"321e17e5-7b25-418a-bf50-f5fb5559000e","year":2025},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:ac43a184f7133c9a364f04078f8280de0412aab4d26f2c587af6f0488087b0a0","observation_id":"54a04a1b-36ca-431f-9008-d69cdccab5b1","resolution":{"observed_at":"2026-07-02T12:06:56.217723Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.01805","last_updated":"2025-05-21T09:38:44Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-02T15:12:17Z","title":"SpaceR: Reinforcing MLLMs in Video Spatial Reasoning","version":2},"cited_work":{"arxiv_id":"2504.01805","doi":null,"metadata_source":"pith","pith_arxiv_id":"2504.01805","snapshot_observed_at":"2026-07-05T11:41:02.483857Z","title":"SpaceR: Reinforcing MLLMs in Video Spatial Reasoning","venue":"cs.CV","work_id":"df3c8f70-c0d0-469a-b74b-b32ceb50d29d","year":2025},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"cited_paper":"/paper/2504.01805","citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:1e8433598e91a76f2b7c70c5b6a323eba79056f757acce36cd30982fce33d01e","observation_id":"841ba6d2-7fd9-4931-8f9b-d3fc0c798296","resolution":{"observed_at":"2026-07-02T12:06:56.178134Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.21776","last_updated":"2025-10-22T16:42:24Z","snapshot_observed_at":"2026-08-05T07:15:29.998948Z","submitted_at":"2025-03-27T17:59:51Z","title":"Video-R1: Reinforcing Video Reasoning in MLLMs","version":4},"cited_work":{"arxiv_id":"2503.21776","doi":"10.48550/arxiv.2503.21776","metadata_source":"pith","pith_arxiv_id":"2503.21776","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Video-R1: Reinforcing Video Reasoning in MLLMs","venue":"cs.CV","work_id":"0ce88332-564c-4361-8e2a-3850eb1ace9c","year":2025},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"cited_paper":"/paper/2503.21776","citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:ab7eb59d463e292fbfac44841a60ca198a8d4abd4e778414de45fc1eb3d2d580","observation_id":"362374fb-70e5-49f5-b2c6-6057ec93d5a0","resolution":{"observed_at":"2026-07-02T12:06:56.211206Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-13T08:50:13.017456+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-13T08:50:13.017456+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02029","last_updated":"2025-09-14T06:49:43Z","snapshot_observed_at":"2026-08-06T20:38:11.691093Z","submitted_at":"2025-07-02T17:05:33Z","title":"RoboBrain 2.0 Technical Report","version":5},"cited_work":{"arxiv_id":"2507.02029","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.02029","snapshot_observed_at":"2026-07-08T12:04:50.478072Z","title":"Robobrain 2.0 technical report","venue":"cs.RO","work_id":"b5af43a7-ae6e-46b6-afdd-0e8e4a3a5da2","year":2025},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"cited_paper":"/paper/2507.02029","citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:604a751836cbe1cd295627dd9bd9e8094e65bbbac24e42692a65808898265533","observation_id":"6e5147c4-8e9e-4803-9239-d81c15592bd9","resolution":{"observed_at":"2026-07-02T12:06:56.199501Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09965","last_updated":"2025-06-19T03:46:55Z","snapshot_observed_at":"2026-07-31T21:40:49.363128Z","submitted_at":"2025-06-11T17:41:50Z","title":"Reinforcing Spatial Reasoning in Vision-Language Models with Interwoven Thinking and Visual Drawing","version":2},"cited_work":{"arxiv_id":"2506.09965","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.09965","snapshot_observed_at":"2026-07-05T11:41:02.594727Z","title":"Reinforcing Spatial Reasoning in Vision-Language Models with Interwoven Thinking and Visual Drawing","venue":"cs.CV","work_id":"ff3bffaa-30ad-4319-b157-10557ee89344","year":2025},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"cited_paper":"/paper/2506.09965","citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:1298d1c189082fc8ad832c6e7f5e7339bcbb45da117c9b9f02f2db3084f0536a","observation_id":"da8a16b1-6bf9-4181-b5e8-a118aa832351","resolution":{"observed_at":"2026-07-02T12:06:56.175437Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.11027","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T03:29:30.017025Z","title":"Vlaser: Vision-language-action model with synergistic embodied reasoning.arXiv preprint arXiv:2510.11027, 2025b","venue":null,"work_id":"967c9002-b47c-4738-86de-c706820dfd2b","year":2025},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:7499a245ecbe8ae6589a4f42438dac307adb4be0c4b4fa310e56a9c98610f6ef","observation_id":"94f8ade3-e80b-48c4-b227-b33d6088f56c","resolution":{"observed_at":"2026-07-02T12:06:56.230089Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T02:25:30.998989Z","title":"Glm-4.1 v-thinking: Towards versatile multimodal reasoning with scalable reinforcement learning.arXiv e-prints, pages arXiv–2507, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:f01688b4cff550140cfb681493c4ae0e0c02d543135b579dab0ffa8f5603f4be","observation_id":"38952c06-3a32-426a-9005-38b9e53cbda3","resolution":{"observed_at":"2026-06-28T02:25:30.998989Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T02:25:30.998989Z","title":"Gpt-4o.https://openai.com/index/hello-gpt-4o/, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:4c90ce0386654fc29e627c5660c6fc3c4dadbfdc759473f64a0a2c0f1a444244","observation_id":"a407419a-5b2f-4f0b-ba0d-ce93ad569cd1","resolution":{"observed_at":"2026-06-28T02:25:30.998989Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T02:25:30.998989Z","title":"Gemini 2.5: Our most intelligent ai model.https://blog.google/technology/google-deepmind/ gemini-model-thinking-updates-march-2025/, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:d2782ed88cf1090ae2c2ce9c56eede912a33564005fbf976c01f427264c15275","observation_id":"20a12e7f-d579-4b2f-8ef9-a9fc09124afd","resolution":{"observed_at":"2026-06-28T02:25:30.998989Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T02:25:30.998989Z","title":"Gemini 3 flash.https://deepmind.google/models/gemini/flash/, December 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:2dfe31df68f80c25349bbc3e6602a58475ecdf09accd3b6e8b0dc6eec61a8c41","observation_id":"ee84493c-0989-4272-a41f-a2ae5e5fc602","resolution":{"observed_at":"2026-06-28T02:25:30.998989Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.05499","last_updated":"2024-07-19T06:00:41Z","snapshot_observed_at":"2026-07-06T15:00:58.804337Z","submitted_at":"2023-03-09T18:52:16Z","title":"Grounding DINO: Marrying DINO with Grounded Pre-Training for Open-Set Object Detection","version":5},"cited_work":{"arxiv_id":"2303.05499","doi":null,"metadata_source":"pith","pith_arxiv_id":"2303.05499","snapshot_observed_at":"2026-07-10T23:17:45.410080Z","title":"Grounding DINO: Marrying DINO with Grounded Pre-Training for Open-Set Object Detection","venue":"cs.CV","work_id":"3757dc8f-79d5-4beb-a03b-eb4c9a33427d","year":2023},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"cited_paper":"/paper/2303.05499","citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:8a9a8b2c2487293aa24e8d081876adf9b5118982346a0f6a499c342b0d10e33f","observation_id":"3cd57005-a286-48f6-a62c-6d9d5a184813","resolution":{"observed_at":"2026-07-02T12:06:56.224787Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2511.13719","doi":"10.48550/arxiv.2511.13719","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Scaling spatial intelligence with multimodal foundation models","venue":"arXiv (Cornell University)","work_id":"ceb84861-8bdb-4822-864c-8e2d0ae4d322","year":2025},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:6256b43896b880fee3901ff6d656c8ebf315a8f6df5820be8af19110a78d2ff8","observation_id":"7eea73cf-a292-41ff-88e9-c43a38a9da3e","resolution":{"observed_at":"2026-07-02T12:06:56.218076Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T02:25:30.998989Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:3d42cb5a61437aee69111ed235e6f58198198caddbf311dfbd2c48e52acfa997","observation_id":"6da994d8-356a-422f-ab97-d7ec6081a762","resolution":{"observed_at":"2026-06-28T02:25:30.998989Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-28T02:25:30.998989Z","title":"move 2.5 meters to the left","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-06-28T02:25:30.998989Z"},"links":{"citing_paper":"/paper/2606.06476"},"observation_digest":"sha256:5e4aaa0d630db93ddc1925b222771b0b209269cb64570135fc48ef77fbb7c0ec","observation_id":"58ceb36c-2bd3-4a54-ae02-4de813d064de","resolution":{"observed_at":"2026-06-28T02:25:30.998989Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2606.06476","last_updated":"2026-06-04T17:56:36Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T23:46:20.554117Z","submitted_at":"2026-06-04T17:56:36Z","title":"Thinking with Imagination: Agentic Visual Spatial Reasoning with World Simulators"},"reference_resolution":{"displayed":52,"state_counts":{"malformed_identifier":0,"metadata_mismatch":3,"parse_uncertain":0,"unresolved":15,"verified_exact":34,"verified_fuzzy":0},"total_outbound_references":52},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 52 of 52 outbound references and 2 inbound Pith citation observations for arXiv:2606.06476."}