{"as_of":"2026-08-07T06:35:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:afd995b80db75b79b2e1f760567071336ad0948b17612d898c6e0a38c27a7b89","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":33,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":33,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":33,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":33,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T00:51:48.968184Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T16:18:37.306944Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":"2406.13642","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-07-03T16:18:37.306944Z","title":"Spatialbot: Precise spatial understanding with vision language models.arXiv preprint arXiv:2406.13642","venue":null,"work_id":"8038b5ef-8748-4dc8-870d-38bbab426489","year":2024},"citing_paper":{"arxiv_id":"2412.14171","last_updated":"2025-07-02T21:00:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-18T18:59:54Z","title":"Thinking in Space: How Multimodal Large Language Models See, Remember, and Recall Spaces","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-22T09:27:43.919941Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2412.14171"},"observation_digest":"sha256:b22ab585fd97f6b72bca2e05932a0f2313e705a8e705cc5eae0e3b6cecb3a927","observation_id":"e251e3d7-f2cf-4e7b-a04c-63d90c83f091","resolution":{"observed_at":"2026-05-22T09:27:44.172737Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-08-07T00:51:24.318444Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12525","last_updated":"2025-06-14T14:52:38Z","snapshot_observed_at":"2026-08-07T00:44:32.497653Z","submitted_at":"2025-06-14T14:52:38Z","title":"A Spatial Relationship Aware Dataset for Robotics","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T00:51:24.318444Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2506.12525"},"observation_digest":"sha256:12fb59bfd016495742946da1d8ad4802cad8022a549b12fd65bfc55435858a60","observation_id":"e726471f-2f61-4b87-ae35-606d52eb03d2","resolution":{"observed_at":"2026-08-07T00:51:24.318444Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-08-07T00:51:48.968184Z","title":"Spatialbot: Precise spatial understanding with vision language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.12610","last_updated":"2025-07-17T00:39:38Z","snapshot_observed_at":"2026-08-07T00:42:55.264522Z","submitted_at":"2025-06-14T19:10:23Z","title":"OscNet v1.5: Energy Efficient Hopfield Network on CMOS Oscillators for Image Classification","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T00:51:48.968184Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2506.12610"},"observation_digest":"sha256:c3e1f753b2041e155377ec9e541e72d377b87dabff624a32324291494b8f49b5","observation_id":"64ec50b4-7244-4a6f-b574-b2061b2a6d40","resolution":{"observed_at":"2026-08-07T00:51:48.968184Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-08-06T23:01:29.887022Z","title":"Spatialbot: Precise spatial understanding with vision lan- guage models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.20066","last_updated":"2025-06-24T23:58:20Z","snapshot_observed_at":"2026-08-06T22:55:09.041478Z","submitted_at":"2025-06-24T23:58:20Z","title":"ToSA: Token Merging with Spatial Awareness","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T23:01:29.887022Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2506.20066"},"observation_digest":"sha256:b9c61b7f9ef04df9cd58adcbf14ede82863c21aa8c2ed94cdb294306c9b47778","observation_id":"ae73add2-f905-443d-b7aa-abbab6e06216","resolution":{"observed_at":"2026-08-06T23:01:29.887022Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-08-06T20:52:02.423508Z","title":"Spatialbot: Precise spatial understanding with vision language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.01634","last_updated":"2025-07-02T12:05:57Z","snapshot_observed_at":"2026-08-06T20:44:20.663929Z","submitted_at":"2025-07-02T12:05:57Z","title":"Depth Anything at Any Condition","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T20:52:02.423508Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2507.01634"},"observation_digest":"sha256:267df768b942b6347e16ec629464c88a3836f948c2f460d11a9f3c32cf61e320","observation_id":"27c51cb1-594a-4ea0-9033-28489f4484f4","resolution":{"observed_at":"2026-08-06T20:52:02.423508Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-08-06T21:22:27.079986Z","title":"Spatialbot: Precise spatial understanding with vision language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02978","last_updated":"2025-07-01T03:05:56Z","snapshot_observed_at":"2026-08-06T21:15:42.038021Z","submitted_at":"2025-07-01T03:05:56Z","title":"Ascending the Infinite Ladder: Benchmarking Spatial Deformation Reasoning in Vision-Language Models","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T21:22:27.079986Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2507.02978"},"observation_digest":"sha256:cc73ee50b676d95e8e594853038802d15f1cac79cfdcf43f1f33b57191828725","observation_id":"5404da64-bf32-4d31-9561-f7a31b58e557","resolution":{"observed_at":"2026-08-06T21:22:27.079986Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-08-06T20:15:52.733114Z","title":"Spatialbot: Precise spatial understanding with vision language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.03483","last_updated":"2025-07-08T05:05:04Z","snapshot_observed_at":"2026-08-06T20:06:25.553030Z","submitted_at":"2025-07-04T11:20:09Z","title":"BMMR: A Large-Scale Bilingual Multimodal Multi-Discipline Reasoning Dataset","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T20:15:52.733114Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2507.03483"},"observation_digest":"sha256:af9ea8e27793dbdab323b60903ae4c8888405a1489f3fb7bf0171dc0e226d23b","observation_id":"c569df7c-8152-4629-8373-70c74568eca3","resolution":{"observed_at":"2026-08-06T20:15:52.733114Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-08-06T20:04:38.185326Z","title":"Spatialbot: Precise spatial understanding with vision lan- guage models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.03930","last_updated":"2025-07-08T01:07:30Z","snapshot_observed_at":"2026-08-06T19:56:53.126988Z","submitted_at":"2025-07-05T07:29:37Z","title":"RwoR: Generating Robot Demonstrations from Human Hand Collection for Policy Learning without Robot","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T20:04:38.185326Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2507.03930"},"observation_digest":"sha256:b360291f0c54613b27d017d1a89aa571579744c16a2f75f976d28e643cb7363a","observation_id":"dbb6d299-169c-4016-a284-3201659f70f7","resolution":{"observed_at":"2026-08-06T20:04:38.185326Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-08-06T19:48:07.793262Z","title":"Spatialbot: Precise spatial understanding with vision language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.04633","last_updated":"2025-07-07T03:28:03Z","snapshot_observed_at":"2026-08-06T19:41:05.819122Z","submitted_at":"2025-07-07T03:28:03Z","title":"PRISM: Pointcloud Reintegrated Inference via Segmentation and Cross-attention for Manipulation","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T19:48:07.793262Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2507.04633"},"observation_digest":"sha256:1525096de6217758d21a2678d5a5de6a39cf4d46704e5e7a4aa65bba262db995","observation_id":"79ba07b9-c5b4-4f00-afe7-84d27c3f0500","resolution":{"observed_at":"2026-08-06T19:48:07.793262Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-08-06T17:29:38.536084Z","title":"Spatialbot: Precise spatial understanding with vision language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.10778","last_updated":"2025-08-14T03:48:03Z","snapshot_observed_at":"2026-08-06T17:23:22.701464Z","submitted_at":"2025-07-14T20:05:55Z","title":"Warehouse Spatial Question Answering with LLM Agent","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T17:29:38.536084Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2507.10778"},"observation_digest":"sha256:7166a598742aa42b832ebafd5f1534c8d45a0a6322d74168faa05b104200aaa6","observation_id":"e2ed1f8c-2a8d-4720-84e0-5316589660b5","resolution":{"observed_at":"2026-08-06T17:29:38.536084Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-08-06T19:54:48.156627Z","title":"Spatialbot: Precise spatial understanding with vision language models,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.13362","last_updated":"2025-07-06T10:51:12Z","snapshot_observed_at":"2026-08-06T19:47:41.467899Z","submitted_at":"2025-07-06T10:51:12Z","title":"Enhancing Spatial Reasoning in Vision-Language Models via Chain-of-Thought Prompting and Reinforcement Learning","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T19:54:48.156627Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2507.13362"},"observation_digest":"sha256:f1cd6f8bb3ef006750b4ee3fc82baccbc73605e993e4397f90e1691517443b1b","observation_id":"46a2075a-2a04-4750-af4a-544bd439c12c","resolution":{"observed_at":"2026-08-06T19:54:48.156627Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-08-06T15:38:59.749275Z","title":"Spatialbot: Precise spatial understanding with vision language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.15321","last_updated":"2025-07-21T07:23:14Z","snapshot_observed_at":"2026-08-06T15:32:39.453149Z","submitted_at":"2025-07-21T07:23:14Z","title":"BenchDepth: Are We on the Right Way to Evaluate Depth Foundation Models?","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T15:38:59.749275Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2507.15321"},"observation_digest":"sha256:5b746721e70643b97a81ef37483f280fe5c1fad31625739b00a91ef56e359fe5","observation_id":"5957f5e9-d878-47cd-9323-840f7158ce10","resolution":{"observed_at":"2026-08-06T15:38:59.749275Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-08-05T22:23:09.481651Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.07135","last_updated":"2025-08-10T01:15:37Z","snapshot_observed_at":"2026-08-05T22:23:08.270280Z","submitted_at":"2025-08-10T01:15:37Z","title":"Canvas3D: Empowering Precise Spatial Control for Image Generation with Constraints from a 3D Virtual Canvas","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-05T22:23:09.481651Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2508.07135"},"observation_digest":"sha256:4d58d7a31aecb9791b6cab2eb7cd9d403762f1c4c46dca9a9ebfd4e293187260","observation_id":"1cfeaa5f-2de8-42bf-b014-002bbc930982","resolution":{"observed_at":"2026-08-05T22:23:09.481651Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":"2406.13642","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-07-03T16:18:37.306944Z","title":"Spatialbot: Precise spatial understanding with vision language models.arXiv preprint arXiv:2406.13642","venue":null,"work_id":"8038b5ef-8748-4dc8-870d-38bbab426489","year":2024},"citing_paper":{"arxiv_id":"2508.13998","last_updated":"2026-04-06T03:29:44Z","snapshot_observed_at":"2026-08-03T11:31:47.613184Z","submitted_at":"2025-08-19T16:50:01Z","title":"Embodied-R1: Reinforced Embodied Reasoning for General Robotic Manipulation","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-18T22:04:34.235731Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2508.13998"},"observation_digest":"sha256:dff6e49eb446aedc65dde2363eaeb691938f2e4d1ed54adc3db63c9a70d04ad3","observation_id":"452ba9fc-3031-4cb0-aa31-3c394b634645","resolution":{"observed_at":"2026-05-18T22:06:51.720215Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-08-05T11:56:07.891723Z","title":"Spatialbot: Precise spatial understanding with vision language models, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.02175","last_updated":"2025-09-04T16:38:44Z","snapshot_observed_at":"2026-08-05T11:56:06.572610Z","submitted_at":"2025-09-02T10:32:58Z","title":"Understanding Space Is Rocket Science -- Only Top Reasoning Models Can Solve Spatial Understanding Tasks","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-05T11:56:07.891723Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2509.02175"},"observation_digest":"sha256:a3f042dd764dc88c15e4d5528c93c879186d0cf7b6f934e99bf30e8930a6071e","observation_id":"69047047-b5e5-45cd-8174-14a6659374bd","resolution":{"observed_at":"2026-08-05T11:56:07.891723Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-08-04T11:38:11.546062Z","title":"Spatialbot: Precise spatial understanding with vision language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2510.03896","last_updated":"2026-06-11T03:17:22Z","snapshot_observed_at":"2026-08-07T01:06:23.705525Z","submitted_at":"2025-10-04T18:33:27Z","title":"GAE: Unleashing Physical Potential of VLM with Generalizable Action Expert","version":2},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-04T11:38:11.546062Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2510.03896"},"observation_digest":"sha256:e8f440b192b2daaa484413574e16fd23a6f6cef0d8dbaaafcf97903d1ac30e89","observation_id":"82cd167f-a1ad-452d-a6d8-9646c13e431c","resolution":{"observed_at":"2026-08-04T11:38:11.546062Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":"2406.13642","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-07-03T16:18:37.306944Z","title":"Spatialbot: Precise spatial understanding with vision language models.arXiv preprint arXiv:2406.13642","venue":null,"work_id":"8038b5ef-8748-4dc8-870d-38bbab426489","year":2024},"citing_paper":{"arxiv_id":"2511.06754","last_updated":"2026-05-06T07:00:01Z","snapshot_observed_at":"2026-07-29T01:21:11.279990Z","submitted_at":"2025-11-10T06:33:44Z","title":"SlotVLA: Towards Modeling of Object-Relation Representations in Robotic Manipulation","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-18T00:22:41.611893Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2511.06754"},"observation_digest":"sha256:29ee31210df3ba74857d15f727010cf94a1e79df6c46f2582fd3285b501a4380","observation_id":"a73e99a7-7552-4f1a-9e21-59d69faf6d4b","resolution":{"observed_at":"2026-05-18T00:25:32.994831Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-08-03T23:08:49.001243Z","title":"Spatialbot: Precise spatial understanding with vision language models.arXiv preprint arXiv:2406.13642,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2511.07403","last_updated":"2026-07-02T19:21:24Z","snapshot_observed_at":"2026-08-06T17:52:09.094066Z","submitted_at":"2025-11-10T18:52:47Z","title":"SpatialThinker: Reinforcing Scene Graph-Grounded Spatial Reasoning via Dense Rewards","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-03T23:08:49.001243Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2511.07403"},"observation_digest":"sha256:858ba6bcef4f6987be9608cf3a0a07599095290e2c970e4ec3b167f8866e9a5c","observation_id":"c89204c0-a66f-433b-9146-90382ced24ee","resolution":{"observed_at":"2026-08-03T23:08:49.001243Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":"2406.13642","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-07-03T16:18:37.306944Z","title":"Spatialbot: Precise spatial understanding with vision language models.arXiv preprint arXiv:2406.13642","venue":null,"work_id":"8038b5ef-8748-4dc8-870d-38bbab426489","year":2024},"citing_paper":{"arxiv_id":"2511.17411","last_updated":"2026-04-27T17:16:04Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-11-21T17:09:43Z","title":"SPEAR-1: Scaling Beyond Robot Demonstrations via 3D Understanding","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-17T20:21:12.375936Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2511.17411"},"observation_digest":"sha256:8572dc83f942872869e363632cf150ce44bca4f58b7ee356075d2d6e234af0cf","observation_id":"f4991661-264f-47e3-b6ef-96f837d1bf93","resolution":{"observed_at":"2026-05-17T20:22:04.620503Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-08-03T20:38:54.925300Z","title":"Spatialbot: Precise spatial understanding with vision language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2511.19119","last_updated":"2026-06-28T11:41:55Z","snapshot_observed_at":"2026-08-05T00:45:17.067470Z","submitted_at":"2025-11-24T13:49:17Z","title":"MonoSR: Open-Vocabulary Spatial Reasoning from Monocular Images","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-03T20:38:54.925300Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2511.19119"},"observation_digest":"sha256:bd2e8c5c5b9d7c9d871a5da09b3add7a94c8a7523ec9966310790eef782ee099","observation_id":"89fc6ba7-bae3-419e-9de8-09a4738a8d4a","resolution":{"observed_at":"2026-08-03T20:38:54.925300Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":"2406.13642","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-07-03T16:18:37.306944Z","title":"Spatialbot: Precise spatial understanding with vision language models.arXiv preprint arXiv:2406.13642","venue":null,"work_id":"8038b5ef-8748-4dc8-870d-38bbab426489","year":2024},"citing_paper":{"arxiv_id":"2604.07592","last_updated":"2026-04-08T20:49:50Z","snapshot_observed_at":"2026-07-06T22:55:46.307881Z","submitted_at":"2026-04-08T20:49:50Z","title":"Spatio-Temporal Grounding of Large Language Models from Perception Streams","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-10T17:10:45.837684Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2604.07592"},"observation_digest":"sha256:e3528b070b98e9bc367033cedd88426754e0e5847f2bb01287bcc3d970da6067","observation_id":"0a578d09-6ef1-47c0-acc8-fb7471255bc4","resolution":{"observed_at":"2026-05-11T07:26:02.678079Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":"2406.13642","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-07-03T16:18:37.306944Z","title":"Spatialbot: Precise spatial understanding with vision language models.arXiv preprint arXiv:2406.13642","venue":null,"work_id":"8038b5ef-8748-4dc8-870d-38bbab426489","year":2024},"citing_paper":{"arxiv_id":"2604.20012","last_updated":"2026-04-21T21:40:58Z","snapshot_observed_at":"2026-07-06T23:06:31.110859Z","submitted_at":"2026-04-21T21:40:58Z","title":"EmbodiedMidtrain: Bridging the Gap between Vision-Language Models and Vision-Language-Action Models via Mid-training","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-10T02:16:08.687340Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2604.20012"},"observation_digest":"sha256:e8298720606efc441de3d25a68ae1bc39a159b35a7bd531b447cae74fd807c64","observation_id":"64bbe8e8-686b-4ab3-8c2a-8a9ecec42f57","resolution":{"observed_at":"2026-05-11T13:11:03.815073Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":"2406.13642","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-07-03T16:18:37.306944Z","title":"Spatialbot: Precise spatial understanding with vision language models.arXiv preprint arXiv:2406.13642","venue":null,"work_id":"8038b5ef-8748-4dc8-870d-38bbab426489","year":2024},"citing_paper":{"arxiv_id":"2605.05997","last_updated":"2026-05-22T12:07:44Z","snapshot_observed_at":"2026-08-02T14:48:34.696358Z","submitted_at":"2026-05-07T10:48:46Z","title":"4DThinker: Thinking with 4D Imagery for Dynamic Spatial Understanding","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-08T14:20:08.404090Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2605.05997"},"observation_digest":"sha256:ef7618139668ab00efcc380100ea76636b4c364abd8ce04bdcee12880633df6c","observation_id":"54209770-f05e-4e37-ba5e-3fc297cfd849","resolution":{"observed_at":"2026-05-11T18:41:11.692955Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":"2406.13642","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-07-03T16:18:37.306944Z","title":"Spatialbot: Precise spatial understanding with vision language models.arXiv preprint arXiv:2406.13642","venue":null,"work_id":"8038b5ef-8748-4dc8-870d-38bbab426489","year":2024},"citing_paper":{"arxiv_id":"2605.05997","last_updated":"2026-05-22T12:07:44Z","snapshot_observed_at":"2026-08-02T14:48:34.696358Z","submitted_at":"2026-05-07T10:48:46Z","title":"4DThinker: Thinking with 4D Imagery for Dynamic Spatial Understanding","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-25T06:15:33.062980Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2605.05997"},"observation_digest":"sha256:b569c6e2ca2839a6099a12f01ca0aee7f124a86f0fd6dc80e73b2abb916e42c5","observation_id":"fa29e92c-af28-48a0-a4ed-1caaaf27aa2f","resolution":{"observed_at":"2026-05-25T06:16:39.992535Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":"2406.13642","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-07-03T16:18:37.306944Z","title":"Spatialbot: Precise spatial understanding with vision language models.arXiv preprint arXiv:2406.13642","venue":null,"work_id":"8038b5ef-8748-4dc8-870d-38bbab426489","year":2024},"citing_paper":{"arxiv_id":"2605.10588","last_updated":"2026-05-11T13:59:09Z","snapshot_observed_at":"2026-07-06T23:22:32.858399Z","submitted_at":"2026-05-11T13:59:09Z","title":"Thinking with Novel Views: A Systematic Analysis of Generative-Augmented Spatial Intelligence","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-05-12T03:24:41.877312Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2605.10588"},"observation_digest":"sha256:ae0756672677447522ea48d890f6ca88af80dac104977686086bb91eb1bea251","observation_id":"9aa7b8c5-3f88-49ab-bdd7-ffe4c9f2a117","resolution":{"observed_at":"2026-05-12T03:26:19.382386Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":"2406.13642","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-07-03T16:18:37.306944Z","title":"Spatialbot: Precise spatial understanding with vision language models.arXiv preprint arXiv:2406.13642","venue":null,"work_id":"8038b5ef-8748-4dc8-870d-38bbab426489","year":2024},"citing_paper":{"arxiv_id":"2605.18746","last_updated":"2026-05-25T08:34:52Z","snapshot_observed_at":"2026-08-02T07:58:31.278661Z","submitted_at":"2026-05-18T17:59:02Z","title":"ESI-Bench: Towards Embodied Spatial Intelligence that Closes the Perception-Action Loop","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-20T10:52:22.778489Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2605.18746"},"observation_digest":"sha256:4e83719562ecc329f85d5ec341e6e18b05c2f69bb4e150da72513de16fa52a7e","observation_id":"88753f51-bdb3-4f4a-892d-3971c0915305","resolution":{"observed_at":"2026-05-20T10:53:13.222326Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":"2406.13642","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-07-03T16:18:37.306944Z","title":"Spatialbot: Precise spatial understanding with vision language models.arXiv preprint arXiv:2406.13642","venue":null,"work_id":"8038b5ef-8748-4dc8-870d-38bbab426489","year":2024},"citing_paper":{"arxiv_id":"2605.18746","last_updated":"2026-05-25T08:34:52Z","snapshot_observed_at":"2026-08-02T07:58:31.278661Z","submitted_at":"2026-05-18T17:59:02Z","title":"ESI-Bench: Towards Embodied Spatial Intelligence that Closes the Perception-Action Loop","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-30T18:25:17.831116Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2605.18746"},"observation_digest":"sha256:8ddf98d8bb8c10af4a8f0ed8c028d65c14492b5b483014d86069400445615afa","observation_id":"b0018adf-4f46-4670-9535-fda813036c2d","resolution":{"observed_at":"2026-07-01T15:05:47.182481Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":"2406.13642","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-07-03T16:18:37.306944Z","title":"Spatialbot: Precise spatial understanding with vision language models.arXiv preprint arXiv:2406.13642","venue":null,"work_id":"8038b5ef-8748-4dc8-870d-38bbab426489","year":2024},"citing_paper":{"arxiv_id":"2605.25979","last_updated":"2026-05-25T15:54:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T15:54:04Z","title":"LLaVA-OneVision-2: Towards Next-Generation Perceptual Intelligence","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-29T22:12:05.365596Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2605.25979"},"observation_digest":"sha256:2b5dabeede9ef918f304dea4a7bbf8c04200733cef3554d177ddca5e03687ef4","observation_id":"016b2ce0-c946-4469-add7-ec1b9b10e22c","resolution":{"observed_at":"2026-06-29T22:13:59.560592Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":"2406.13642","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-07-03T16:18:37.306944Z","title":"Spatialbot: Precise spatial understanding with vision language models.arXiv preprint arXiv:2406.13642","venue":null,"work_id":"8038b5ef-8748-4dc8-870d-38bbab426489","year":2024},"citing_paper":{"arxiv_id":"2605.30561","last_updated":"2026-05-28T20:48:55Z","snapshot_observed_at":"2026-07-06T23:39:48.531480Z","submitted_at":"2026-05-28T20:48:55Z","title":"VLM3: Vision Language Models Are Native 3D Learners","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-29T07:45:31.978215Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2605.30561"},"observation_digest":"sha256:e2f3284e4063b49a0cb80349504f641b993fcf04a30544326abc81f62dc1d765","observation_id":"effb4d79-bfe3-4ed5-b356-8d1378a6ecba","resolution":{"observed_at":"2026-06-29T07:53:14.019914Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":"2406.13642","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-07-03T16:18:37.306944Z","title":"Spatialbot: Precise spatial understanding with vision language models.arXiv preprint arXiv:2406.13642","venue":null,"work_id":"8038b5ef-8748-4dc8-870d-38bbab426489","year":2024},"citing_paper":{"arxiv_id":"2606.05445","last_updated":"2026-06-03T21:08:06Z","snapshot_observed_at":"2026-08-06T20:42:38.181639Z","submitted_at":"2026-06-03T21:08:06Z","title":"Brick-Composer: Using MLLMs for Assembly with Diverse Bricks","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-28T05:59:29.302038Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2606.05445"},"observation_digest":"sha256:b2b1ade405cc4a73c2d095fc194b374ab7046c6e1b0867d79c71942f2d6ef6f6","observation_id":"6f97f28b-55ff-41db-b7ee-b719312b5659","resolution":{"observed_at":"2026-07-02T08:26:48.424339Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":"2406.13642","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-07-03T16:18:37.306944Z","title":"Spatialbot: Precise spatial understanding with vision language models.arXiv preprint arXiv:2406.13642","venue":null,"work_id":"8038b5ef-8748-4dc8-870d-38bbab426489","year":2024},"citing_paper":{"arxiv_id":"2606.31257","last_updated":"2026-06-30T07:33:18Z","snapshot_observed_at":"2026-07-07T00:04:58.486133Z","submitted_at":"2026-06-30T07:33:18Z","title":"Decodable Is Not Grounded: A Vision-Ablation Arbiter for VLM Spatial Reasoning","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-07-01T06:23:00.372251Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2606.31257"},"observation_digest":"sha256:a3e7a3df3e8a3c927fd815df40da706e50b7a905f9004531413f27fc24448bee","observation_id":"443ec8bd-ee16-464b-b4fb-1deb9e64acf5","resolution":{"observed_at":"2026-07-01T09:35:41.110274Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":"2406.13642","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-07-03T16:18:37.306944Z","title":"Spatialbot: Precise spatial understanding with vision language models.arXiv preprint arXiv:2406.13642","venue":null,"work_id":"8038b5ef-8748-4dc8-870d-38bbab426489","year":2024},"citing_paper":{"arxiv_id":"2607.01784","last_updated":"2026-07-02T06:56:29Z","snapshot_observed_at":"2026-08-03T11:23:56.400674Z","submitted_at":"2026-07-02T06:56:29Z","title":"SpaceEra++: A Unified Framework Towards 3D Spatial Reasoning in Video","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-07-03T16:16:41.412451Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2607.01784"},"observation_digest":"sha256:079edaee4a58d2f5b2e61041fab9ec99567d42f7590e4b90235ec91afa451eec","observation_id":"bb376c85-f6f5-470b-b01f-51d3b0834949","resolution":{"observed_at":"2026-07-03T16:18:37.308596Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.13642","snapshot_observed_at":"2026-08-06T21:49:20.587662Z","title":"arXiv preprint arXiv:2406.13642 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.04574","last_updated":"2026-08-05T08:04:26Z","snapshot_observed_at":"2026-08-07T06:29:56.936289Z","submitted_at":"2026-08-05T08:04:26Z","title":"When Memory Lies: An Empirical Study of Spatial Memory Staleness in VLM Agents","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-06T21:49:20.587662Z"},"links":{"cited_paper":"/paper/2406.13642","citing_paper":"/paper/2608.04574"},"observation_digest":"sha256:b01edcf8ce64b9aeca4bc26fa131010c152c727c30e4565baea22ebd55e3c2a5","observation_id":"83309848-96fa-4915-a231-5899935daab2","resolution":{"observed_at":"2026-08-06T21:49:20.587662Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2406.13642/citation-record","integrity":"/paper/2406.13642/integrity","json":"/paper/2406.13642/citation-record.json","paper":"/paper/2406.13642"},"outbound":[],"paper":{"arxiv_id":"2406.13642","last_updated":"2025-03-19T05:09:14Z","latest_version":7,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-05T18:02:56.273979Z","submitted_at":"2024-06-19T15:41:30Z","title":"SpatialBot: Precise Spatial Understanding with Vision Language Models"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 33 inbound Pith citation observations for arXiv:2406.13642."}