{"as_of":"2026-08-21T10:21:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:3f0e6dc7addce95d43f3044763865eaf4328a67e145b47c1377b0964fbce031f","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":31,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":31,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-21T06:32:19.484+00:00","state":"measured"},{"denominator":31,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":31,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T15:24:47.666879Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-10T09:37:00.841150Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2503.19707","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-07-10T09:37:00.841150Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707","venue":"cs.CV","work_id":"06006849-24f6-4955-9dd0-1cbe59b0fc66","year":2025},"citing_paper":{"arxiv_id":"2505.17012","last_updated":"2026-04-13T12:33:41Z","snapshot_observed_at":"2026-07-06T21:28:43.610849Z","submitted_at":"2025-05-22T17:59:03Z","title":"SpatialScore: Towards Comprehensive Evaluation for Spatial Intelligence","version":3},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-05-22T13:07:11.548885Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2505.17012"},"observation_digest":"sha256:1b5ce35343f953d9749a9ae94e7af962c3dc72b60e5a88e9b0f04d70e4b77dfe","observation_id":"e770b902-a529-42f7-91e2-4410b4695509","resolution":{"observed_at":"2026-05-22T13:11:35.751741Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-08-07T13:50:33.010530Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.20728","last_updated":"2025-08-26T03:25:38Z","snapshot_observed_at":"2026-08-19T23:44:25.543802Z","submitted_at":"2025-05-27T05:17:41Z","title":"Jigsaw-Puzzles: From Seeing to Understanding to Reasoning in Vision-Language Models","version":4},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-07T13:50:33.010530Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2505.20728"},"observation_digest":"sha256:939a7b8e70fc7426dcd655115d628e6b6d31ad7cd8ab74894a00610ca73b5210","observation_id":"fc5d705f-b0e8-4de9-a506-d73312103c29","resolution":{"observed_at":"2026-08-07T13:50:33.010530Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-08-07T10:35:43.691323Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.05440","last_updated":"2025-06-05T12:43:10Z","snapshot_observed_at":"2026-08-07T21:22:23.284742Z","submitted_at":"2025-06-05T12:43:10Z","title":"BYO-Eval: Build Your Own Dataset for Fine-Grained Visual Assessment of Multimodal Language Models","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T10:35:43.691323Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2506.05440"},"observation_digest":"sha256:1ba82b36724834f1078a097d240e938950efa596755a69ac9950e2f38876a75d","observation_id":"0cd3df1a-b09a-4cb4-b3a5-a6eaa402bac7","resolution":{"observed_at":"2026-08-07T10:35:43.691323Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-08-07T00:57:27.247850Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.12374","last_updated":"2025-06-24T10:01:18Z","snapshot_observed_at":"2026-08-16T05:58:22.870564Z","submitted_at":"2025-06-14T07:11:44Z","title":"AntiGrounding: Lifting Robotic Actions into VLM Representation Space for Decision Making","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T00:57:27.247850Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2506.12374"},"observation_digest":"sha256:921f1c49aa2d5c9bbb057cba9b931aebc1b93fa81fb439f38043719a4ccf649b","observation_id":"6e20543d-3d2d-4381-8c77-b14dd9b70932","resolution":{"observed_at":"2026-08-07T00:57:27.247850Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-08-05T15:17:54.990372Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2508.20068","last_updated":"2025-08-27T17:22:34Z","snapshot_observed_at":"2026-08-13T23:17:38.070295Z","submitted_at":"2025-08-27T17:22:34Z","title":"11Plus-Bench: Demystifying Multimodal LLM Spatial Reasoning with Cognitive-Inspired Analysis","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-05T15:17:54.990372Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2508.20068"},"observation_digest":"sha256:d5a7a22d9dc1bc88514e20e2681f28c718f9a1cf652f496c44c0205b00e7fc57","observation_id":"f1ff9a6f-f2be-416f-a1c5-4836384f0f1a","resolution":{"observed_at":"2026-08-05T15:17:54.990372Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-08-03T16:55:22.331240Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2512.11393","last_updated":"2026-08-13T11:40:44Z","snapshot_observed_at":"2026-08-16T23:11:50.762075Z","submitted_at":"2025-12-12T09:07:21Z","title":"The N-Body Problem: Parallel Execution from Single-Person Egocentric Video","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-03T16:55:22.331240Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2512.11393"},"observation_digest":"sha256:65dd7e32aec7641a5eaedcade3c30dfb6d280879818dfe1624f61c2b1f1d707c","observation_id":"e2578177-12ce-4060-8b52-e60f60d175a6","resolution":{"observed_at":"2026-08-03T16:55:22.331240Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-08-03T16:30:36.756028Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2512.13517","last_updated":"2026-05-28T15:09:34Z","snapshot_observed_at":"2026-08-14T14:17:46.093839Z","submitted_at":"2025-12-15T16:43:50Z","title":"A Deep Learning Model of Mental Rotation Informed by Interactive VR Experiments","version":2},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-03T16:30:36.756028Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2512.13517"},"observation_digest":"sha256:713aced1309a265489e5d55b6bec076e8655f8f8a5d039da215ab31724e7fe63","observation_id":"7b3245e6-35e9-49b4-9611-369e9c34a280","resolution":{"observed_at":"2026-08-03T16:30:36.756028Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2503.19707","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-07-10T09:37:00.841150Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707","venue":"cs.CV","work_id":"06006849-24f6-4955-9dd0-1cbe59b0fc66","year":2025},"citing_paper":{"arxiv_id":"2602.11635","last_updated":"2026-04-08T16:46:28Z","snapshot_observed_at":"2026-08-20T04:34:30.705568Z","submitted_at":"2026-02-12T06:37:55Z","title":"Do MLLMs Really Understand Space? A Mathematical Reasoning Evaluation","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-16T03:39:28.364183Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2602.11635"},"observation_digest":"sha256:6b5ad657df24a81fbf5731da5e35987a593eebe9f82f44b839e1fd576d84f53c","observation_id":"f20761b5-2bc6-43be-86e5-777dedd3d6af","resolution":{"observed_at":"2026-05-16T03:40:33.119630Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-07-15T13:03:40.459461Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2603.08011","last_updated":"2026-05-23T08:33:03Z","snapshot_observed_at":"2026-08-17T13:06:45.716705Z","submitted_at":"2026-03-09T06:33:49Z","title":"It's Time to Get It Right: Improving Analog Clock Reading and Clock-Hand Spatial Reasoning in Vision-Language Models","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-07-15T13:03:40.459461Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2603.08011"},"observation_digest":"sha256:e0a19664114cc70f8ab3065af659c40f621599bf82b652a89e9afa3760207914","observation_id":"35b1f545-9ce4-4c8f-8068-6e79edf83bff","resolution":{"observed_at":"2026-07-15T13:03:40.459461Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2503.19707","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-07-10T09:37:00.841150Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707","venue":"cs.CV","work_id":"06006849-24f6-4955-9dd0-1cbe59b0fc66","year":2025},"citing_paper":{"arxiv_id":"2604.15184","last_updated":"2026-04-27T16:57:52Z","snapshot_observed_at":"2026-08-10T22:05:32.806329Z","submitted_at":"2026-04-16T16:15:23Z","title":"Agent-Aided Design for Dynamic CAD Models","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-10T11:13:01.515933Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2604.15184"},"observation_digest":"sha256:1959f581b5e2c6ffd9012b712290aff470a41899c683a0706d54421d980d9055","observation_id":"684a2e35-f9b4-4673-823d-6c853d42bafb","resolution":{"observed_at":"2026-05-10T11:15:10.441699Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2503.19707","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-07-10T09:37:00.841150Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707","venue":"cs.CV","work_id":"06006849-24f6-4955-9dd0-1cbe59b0fc66","year":2025},"citing_paper":{"arxiv_id":"2604.16022","last_updated":"2026-04-17T12:51:46Z","snapshot_observed_at":"2026-08-13T02:15:13.699157Z","submitted_at":"2026-04-17T12:51:46Z","title":"SocialGrid: A Benchmark for Planning and Social Reasoning in Embodied Multi-Agent Systems","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-05-10T08:45:54.303143Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2604.16022"},"observation_digest":"sha256:f0201e18bb8a14067649d7a76ed68da0cca5ae035c9ad352453b7b15bfdd363c","observation_id":"276ae1ac-3d73-47c1-8eeb-35a9d0672089","resolution":{"observed_at":"2026-05-10T08:48:01.663869Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2503.19707","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-07-10T09:37:00.841150Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707","venue":"cs.CV","work_id":"06006849-24f6-4955-9dd0-1cbe59b0fc66","year":2025},"citing_paper":{"arxiv_id":"2604.26614","last_updated":"2026-04-29T12:41:39Z","snapshot_observed_at":"2026-08-11T13:51:24.959545Z","submitted_at":"2026-04-29T12:41:39Z","title":"State Beyond Appearance: Diagnosing and Improving State Consistency in Dial-Based Measurement Reading","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-07T11:45:57.291112Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2604.26614"},"observation_digest":"sha256:258ad573ad22158a452c1d3bf2fb92a5ad5e0ea68db52be39fe84af57494ea87","observation_id":"958a3f9d-f2a8-4c50-9831-c97aa925b0de","resolution":{"observed_at":"2026-05-12T09:16:26.452568Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2503.19707","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-07-10T09:37:00.841150Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707","venue":"cs.CV","work_id":"06006849-24f6-4955-9dd0-1cbe59b0fc66","year":2025},"citing_paper":{"arxiv_id":"2605.09449","last_updated":"2026-05-10T10:01:57Z","snapshot_observed_at":"2026-08-14T01:42:23.694725Z","submitted_at":"2026-05-10T10:01:57Z","title":"SpaceMind++: Toward Allocentric Cognitive Maps for Spatially Grounded Video MLLMs","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-12T04:43:51.900721Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2605.09449"},"observation_digest":"sha256:74fc0ea492004740fd2c805d221d0cd164f95c52c43364f6cb08738c8651df28","observation_id":"2d20f7e4-2d5f-43a6-9df0-955f61c6ee48","resolution":{"observed_at":"2026-05-12T05:56:45.651024Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2503.19707","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-07-10T09:37:00.841150Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707","venue":"cs.CV","work_id":"06006849-24f6-4955-9dd0-1cbe59b0fc66","year":2025},"citing_paper":{"arxiv_id":"2605.09693","last_updated":"2026-05-10T18:25:52Z","snapshot_observed_at":"2026-08-11T14:36:51.006168Z","submitted_at":"2026-05-10T18:25:52Z","title":"Do multimodal models imagine electric sheep?","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-12T03:24:01.933339Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2605.09693"},"observation_digest":"sha256:a55fa7c98bca578fbe4086429be51072ac48e843d62f3f4f096583acfedc224e","observation_id":"8263d4c3-7d96-4686-bf6a-5eedab9bdb88","resolution":{"observed_at":"2026-05-12T03:26:19.535821Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2503.19707","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-07-10T09:37:00.841150Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707","venue":"cs.CV","work_id":"06006849-24f6-4955-9dd0-1cbe59b0fc66","year":2025},"citing_paper":{"arxiv_id":"2605.13169","last_updated":"2026-05-15T16:50:42Z","snapshot_observed_at":"2026-08-11T16:40:12.299909Z","submitted_at":"2026-05-13T08:31:22Z","title":"PanoWorld: Towards Spatial Supersensing in 360$^\\circ$ Panorama World","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-14T20:40:59.877854Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2605.13169"},"observation_digest":"sha256:a3a4e98f65659e5e84007138b65a933583928c5a6cca497073e7a1f9363ceefb","observation_id":"82189804-e2b1-4d7a-9bb5-05fbf4032280","resolution":{"observed_at":"2026-05-14T20:42:57.648243Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2503.19707","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-07-10T09:37:00.841150Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707","venue":"cs.CV","work_id":"06006849-24f6-4955-9dd0-1cbe59b0fc66","year":2025},"citing_paper":{"arxiv_id":"2605.13169","last_updated":"2026-05-15T16:50:42Z","snapshot_observed_at":"2026-08-11T16:40:12.299909Z","submitted_at":"2026-05-13T08:31:22Z","title":"PanoWorld: Towards Spatial Supersensing in 360$^\\circ$ Panorama World","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-19T16:57:03.172340Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2605.13169"},"observation_digest":"sha256:eb4d6c43fb5c9c177ec53c4a09692dafe2dea3fecba84414bd13ab8c0563c9c6","observation_id":"4008e5b3-43d7-4822-adda-ec54db1b1db0","resolution":{"observed_at":"2026-05-19T16:57:40.176444Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2503.19707","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-07-10T09:37:00.841150Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707","venue":"cs.CV","work_id":"06006849-24f6-4955-9dd0-1cbe59b0fc66","year":2025},"citing_paper":{"arxiv_id":"2605.20448","last_updated":"2026-06-18T10:11:04Z","snapshot_observed_at":"2026-08-19T14:13:56.576927Z","submitted_at":"2026-05-19T20:01:19Z","title":"Do Vision-Language Models Understand 3D Scenes or Just Catalogue Objects?","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-21T06:55:04.657347Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2605.20448"},"observation_digest":"sha256:eddf2203e29dd75dfa248a5382c3f78d9c9fb285f8ba4189f100bed123b6a8da","observation_id":"ebe4fb80-1f1c-49cd-835c-d0a57767b29e","resolution":{"observed_at":"2026-05-21T06:59:45.742535Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2503.19707","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-07-10T09:37:00.841150Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707","venue":"cs.CV","work_id":"06006849-24f6-4955-9dd0-1cbe59b0fc66","year":2025},"citing_paper":{"arxiv_id":"2605.20448","last_updated":"2026-06-18T10:11:04Z","snapshot_observed_at":"2026-08-19T14:13:56.576927Z","submitted_at":"2026-05-19T20:01:19Z","title":"Do Vision-Language Models Understand 3D Scenes or Just Catalogue Objects?","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-06-30T17:52:26.785086Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2605.20448"},"observation_digest":"sha256:57cde06f172b65e1fab0c16a81824236ff2b670d1db44f290ffd109835860110","observation_id":"64dbe084-5d50-408c-acab-981e84e68547","resolution":{"observed_at":"2026-06-30T17:54:57.791672Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2503.19707","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-07-10T09:37:00.841150Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707","venue":"cs.CV","work_id":"06006849-24f6-4955-9dd0-1cbe59b0fc66","year":2025},"citing_paper":{"arxiv_id":"2605.23141","last_updated":"2026-05-22T01:43:32Z","snapshot_observed_at":"2026-08-18T17:14:48.084695Z","submitted_at":"2026-05-22T01:43:32Z","title":"VisAnalog: A Diagnostic Suite for Visual Concept Transfer on Natural Images","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-25T05:17:45.034348Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2605.23141"},"observation_digest":"sha256:dc1e8edbe173f3f75b7d211a1683a09a25f80cd215d1ec15258a3b0cb2b53f7b","observation_id":"2730b754-ea6a-4b0e-8042-3c4808aaee64","resolution":{"observed_at":"2026-05-25T05:20:24.981533Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2503.19707","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-07-10T09:37:00.841150Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707","venue":"cs.CV","work_id":"06006849-24f6-4955-9dd0-1cbe59b0fc66","year":2025},"citing_paper":{"arxiv_id":"2605.23176","last_updated":"2026-06-15T18:43:33Z","snapshot_observed_at":"2026-08-18T11:09:47.441160Z","submitted_at":"2026-05-22T02:52:06Z","title":"DRIVESPATIAL: A Benchmark for Spatiotemporal Intelligence in VLMs for Autonomous Driving","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-05-25T05:10:32.522453Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2605.23176"},"observation_digest":"sha256:2e189472ac7ea3a2724bcc2f36283022bda14cb746ab7afd90da5b7214bba117","observation_id":"7577280a-8cca-43d9-a91f-8c3dab8ac721","resolution":{"observed_at":"2026-05-25T05:15:22.790290Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2503.19707","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-07-10T09:37:00.841150Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707","venue":"cs.CV","work_id":"06006849-24f6-4955-9dd0-1cbe59b0fc66","year":2025},"citing_paper":{"arxiv_id":"2605.23176","last_updated":"2026-06-15T18:43:33Z","snapshot_observed_at":"2026-08-18T11:09:47.441160Z","submitted_at":"2026-05-22T02:52:06Z","title":"DRIVESPATIAL: A Benchmark for Spatiotemporal Intelligence in VLMs for Autonomous Driving","version":2},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-06-30T16:40:22.441025Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2605.23176"},"observation_digest":"sha256:ec66870e125be270f30af18af1299940e48bf16d9c6e2dc55a77e1605f24a259","observation_id":"d18c83e4-783f-4d0f-898b-e86c4781003e","resolution":{"observed_at":"2026-06-30T16:44:56.045464Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2503.19707","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-07-10T09:37:00.841150Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707","venue":"cs.CV","work_id":"06006849-24f6-4955-9dd0-1cbe59b0fc66","year":2025},"citing_paper":{"arxiv_id":"2605.23771","last_updated":"2026-05-22T15:40:52Z","snapshot_observed_at":"2026-08-14T17:01:43.585333Z","submitted_at":"2026-05-22T15:40:52Z","title":"PhotoFlow: Agentic 3D Virtual Photography Missions","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-25T04:33:24.355622Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2605.23771"},"observation_digest":"sha256:0ea921245e09db1a90453327bbf82f25aad76c84f92dffcff3d3892406b30e45","observation_id":"c0bcbcf9-3c93-4616-9b86-1034cf286b57","resolution":{"observed_at":"2026-05-25T04:35:21.022926Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2503.19707","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-07-10T09:37:00.841150Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707","venue":"cs.CV","work_id":"06006849-24f6-4955-9dd0-1cbe59b0fc66","year":2025},"citing_paper":{"arxiv_id":"2605.30557","last_updated":"2026-05-28T20:44:47Z","snapshot_observed_at":"2026-08-15T13:58:33.370405Z","submitted_at":"2026-05-28T20:44:47Z","title":"Seeing Isn't Knowing: Do VLMs Know When Not to Answer Spatial Questions (and Why)?","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-06-29T07:48:19.295578Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2605.30557"},"observation_digest":"sha256:e9891c21d0823b7e318c125d482e8a8b5fd9778d55b55062af3e39e28797f973","observation_id":"27886b22-3ddf-456c-9d55-27e77cd0e545","resolution":{"observed_at":"2026-06-29T07:53:13.577911Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2503.19707","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-07-10T09:37:00.841150Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707","venue":"cs.CV","work_id":"06006849-24f6-4955-9dd0-1cbe59b0fc66","year":2025},"citing_paper":{"arxiv_id":"2606.07641","last_updated":"2026-06-01T14:16:09Z","snapshot_observed_at":"2026-08-11T19:58:18.830064Z","submitted_at":"2026-06-01T14:16:09Z","title":"Readable Yet Unpredictable: Rotated-Outcome Prediction in Vision-Language Models","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-28T15:21:16.342973Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2606.07641"},"observation_digest":"sha256:0dcfb4cfffbe9dafe9ba5b4103cce2cf3c8d114989db31678bddda4ebdde1604","observation_id":"63fd505f-af17-4c8f-b2e4-3dc7eb652d1b","resolution":{"observed_at":"2026-07-01T22:26:18.105066Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2503.19707","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-07-10T09:37:00.841150Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707","venue":"cs.CV","work_id":"06006849-24f6-4955-9dd0-1cbe59b0fc66","year":2025},"citing_paper":{"arxiv_id":"2606.11918","last_updated":"2026-06-17T09:46:25Z","snapshot_observed_at":"2026-08-15T04:29:50.319316Z","submitted_at":"2026-06-10T10:50:06Z","title":"The Art of Interrogation: Consistency Amplifies Factuality in Spatial Reasoning","version":2},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-06-27T09:46:16.088490Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2606.11918"},"observation_digest":"sha256:6032bc61ec1c5f16967e8459a3a9fb6fd8d5e3dd8960870496e9ab7c4992182f","observation_id":"68fdc119-0b54-4194-b323-bcbc6f4668fc","resolution":{"observed_at":"2026-07-03T10:58:03.046297Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2503.19707","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-07-10T09:37:00.841150Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707","venue":"cs.CV","work_id":"06006849-24f6-4955-9dd0-1cbe59b0fc66","year":2025},"citing_paper":{"arxiv_id":"2606.30378","last_updated":"2026-06-29T14:38:20Z","snapshot_observed_at":"2026-08-17T11:16:23.738867Z","submitted_at":"2026-06-29T14:38:20Z","title":"OmniCoT: A Benchmark for Global and Multi-Step Panoramic Reasoning","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-06-30T06:11:09.693576Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2606.30378"},"observation_digest":"sha256:f11c5b55841e9d5f0b9db2b4b17fc93ea562be3859024b0f0073a2e5ae2e1fe0","observation_id":"75c0083b-0814-4336-b217-bedf8ab6c357","resolution":{"observed_at":"2026-06-30T06:14:19.130332Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2503.19707","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-07-10T09:37:00.841150Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707","venue":"cs.CV","work_id":"06006849-24f6-4955-9dd0-1cbe59b0fc66","year":2025},"citing_paper":{"arxiv_id":"2606.31467","last_updated":"2026-06-30T10:46:23Z","snapshot_observed_at":"2026-07-07T00:05:12.546815Z","submitted_at":"2026-06-30T10:46:23Z","title":"AeroVerse-SatAgent: UAV-Satellite Collaborative Spatial Reasoning Inspired by the Dual Visual Pathway Theory of Cognitive Neuroscience","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-07-01T06:18:35.494240Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2606.31467"},"observation_digest":"sha256:ddd0bd172847a715a68fc734bde71c8ed6b2b0204a254237a07d62e1d66e28a1","observation_id":"706e41e4-1ee3-4f9c-85a8-d07ae0816756","resolution":{"observed_at":"2026-07-01T09:45:39.862470Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2503.19707","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-07-10T09:37:00.841150Z","title":"Mind the gap: Benchmarking spatial reasoning in vision-language models.arXiv preprint arXiv:2503.19707","venue":"cs.CV","work_id":"06006849-24f6-4955-9dd0-1cbe59b0fc66","year":2025},"citing_paper":{"arxiv_id":"2607.08317","last_updated":"2026-07-09T09:56:50Z","snapshot_observed_at":"2026-08-07T21:53:44.151713Z","submitted_at":"2026-07-09T09:56:50Z","title":"Blind-Spots-Bench: Evaluating Blind Spots in Multimodal Models","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-07-10T09:30:46.564013Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2607.08317"},"observation_digest":"sha256:a4ab9b9142aa9a331cb53ee55adf6003b5a8954e476989b23a506066a7a872d0","observation_id":"a6d702a7-d72f-4393-af86-8a1340ef24b8","resolution":{"observed_at":"2026-07-10T09:37:00.842655Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-08-01T17:25:10.666535Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.17657","last_updated":"2026-07-20T08:07:22Z","snapshot_observed_at":"2026-08-17T23:04:03.272648Z","submitted_at":"2026-07-20T08:07:22Z","title":"OrientSAM: Mitigating Camera-Centric Shortcut in Multimodal Spatial Reasoning via Orientation-Aware Spatial Alignment","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-01T17:25:10.666535Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2607.17657"},"observation_digest":"sha256:ca6d6fbe489558971033511f4f10fcf2ac7b42517a9da052230309c07bf4f338","observation_id":"a2f61674-1d1c-4f5f-9868-4bde23c20bfe","resolution":{"observed_at":"2026-08-01T17:25:10.666535Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-08-01T08:39:39.509832Z","title":"Huajie Tan, Enshen Zhou, Zhiyu Li, Yijie Xu, Yuheng Ji, Xiansheng Chen, Cheng Chi, Pengwei Wang, Huizhu Jia, Yulong Ao, et al","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21072","last_updated":"2026-07-23T09:04:48Z","snapshot_observed_at":"2026-08-19T14:09:20.412548Z","submitted_at":"2026-07-23T09:04:48Z","title":"Show, Don't Tell: Evaluating Spatial Cognition in Generative Pixels Rather Than LLM Text","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-01T08:39:39.509832Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2607.21072"},"observation_digest":"sha256:9299b5c2430aa3e23583cf23a1d9adb336134ffbbe7956da9daeaf2eb9d05bbb","observation_id":"ce3e6dab-3e1f-4d8f-a056-01001d16d180","resolution":{"observed_at":"2026-08-01T08:39:39.509832Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.19707","snapshot_observed_at":"2026-08-15T15:24:47.666879Z","title":"arXiv preprint arXiv:2503.19707 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00726","last_updated":"2026-08-01T15:47:52Z","snapshot_observed_at":"2026-08-17T22:49:17.630687Z","submitted_at":"2026-08-01T15:47:52Z","title":"Foveated Probes Recover Localized Binding Information in Vision Foundation Models","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-15T15:24:47.666879Z"},"links":{"cited_paper":"/paper/2503.19707","citing_paper":"/paper/2608.00726"},"observation_digest":"sha256:cf5d79261ed9c31ac198974a467ecb41e11f66bde0331592abdc5d529443938d","observation_id":"591a34d0-2c22-4a3f-8869-c2f4ae3232e4","resolution":{"observed_at":"2026-08-15T15:24:47.666879Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2503.19707/citation-record","integrity":"/paper/2503.19707/integrity","json":"/paper/2503.19707/citation-record.json","paper":"/paper/2503.19707"},"outbound":[],"paper":{"arxiv_id":"2503.19707","last_updated":"2025-03-25T14:34:06Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-16T12:46:58.006510Z","submitted_at":"2025-03-25T14:34:06Z","title":"Mind the Gap: Benchmarking Spatial Reasoning in Vision-Language Models"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"thesis":"As of 21 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 31 inbound Pith citation observations for arXiv:2503.19707."}