{"as_of":"2026-08-16T18:02:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:7f4b174264d50a1e8c247a6e5bcd1dc425da0a02c71408a72cec67bfbc1a5b00","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":12,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":12,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-16T06:30:59.297886+00:00","state":"measured"},{"denominator":12,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":12,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T11:37:03.205845Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-02T21:07:24.216562Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2410.17385","last_updated":"2025-04-17T17:59:27Z","snapshot_observed_at":"2026-08-16T13:06:51.567111Z","submitted_at":"2024-10-22T19:39:15Z","title":"Do Vision-Language Models Represent Space and How? Evaluating Spatial Frame of Reference Under Ambiguities","version":2},"cited_work":{"arxiv_id":"2410.17385","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2410.17385","snapshot_observed_at":"2026-07-02T21:07:24.216562Z","title":"Do vision-language models represent space and how? eval- uating spatial frame of reference under ambiguities","venue":null,"work_id":"f53c4559-e7da-40a9-935f-1e53b931b6be","year":2024},"citing_paper":{"arxiv_id":"2503.07557","last_updated":"2026-05-02T15:43:17Z","snapshot_observed_at":"2026-08-03T10:34:14.204884Z","submitted_at":"2025-03-10T17:27:17Z","title":"AutoSpatial: Visual-Language Reasoning for Social Robot Navigation through Efficient Spatial Reasoning Learning","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-23T00:26:58.273861Z"},"links":{"cited_paper":"/paper/2410.17385","citing_paper":"/paper/2503.07557"},"observation_digest":"sha256:54c7cad1c41d75abcb12af3c8694cc5869ed988c65653be3c59d4d070c00e705","observation_id":"e58a5a0d-5842-4afd-bd71-a9b4930b7537","resolution":{"observed_at":"2026-05-23T00:27:17.808374Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.17385","last_updated":"2025-04-17T17:59:27Z","snapshot_observed_at":"2026-08-16T13:06:51.567111Z","submitted_at":"2024-10-22T19:39:15Z","title":"Do Vision-Language Models Represent Space and How? Evaluating Spatial Frame of Reference Under Ambiguities","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.17385","snapshot_observed_at":"2026-08-16T11:37:03.205845Z","title":"Do vision-language models represent space and how? evaluating spatial frame of reference under ambiguities.arXiv preprint arXiv:2410.17385, 2024","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.15037","last_updated":"2025-06-03T08:29:23Z","snapshot_observed_at":"2026-08-16T11:32:38.450516Z","submitted_at":"2025-04-21T11:48:39Z","title":"Scaling and Beyond: Advancing Spatial Reasoning in MLLMs Requires New Recipes","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-16T11:37:03.205845Z"},"links":{"cited_paper":"/paper/2410.17385","citing_paper":"/paper/2504.15037"},"observation_digest":"sha256:876e22ef13b76b51d50a355ccd05372079f3e93ee8deaa6ebfd959fe1fe23e5b","observation_id":"1697cb18-7585-47e4-ae01-aecf876f8eac","resolution":{"observed_at":"2026-08-16T11:37:03.205845Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.17385","last_updated":"2025-04-17T17:59:27Z","snapshot_observed_at":"2026-08-16T13:06:51.567111Z","submitted_at":"2024-10-22T19:39:15Z","title":"Do Vision-Language Models Represent Space and How? Evaluating Spatial Frame of Reference Under Ambiguities","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.17385","snapshot_observed_at":"2026-08-16T11:17:06.512196Z","title":"the [blank] is to the left of the [blank]","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.16061","last_updated":"2025-04-22T17:38:01Z","snapshot_observed_at":"2026-08-16T11:09:04.818659Z","submitted_at":"2025-04-22T17:38:01Z","title":"Vision language models are unreliable at trivial spatial cognition","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-16T11:17:06.512196Z"},"links":{"cited_paper":"/paper/2410.17385","citing_paper":"/paper/2504.16061"},"observation_digest":"sha256:26a7b0afcaae86fcba117bc63a49e5050e99ab4eef666e1bc144261458507a98","observation_id":"ec4b1981-974c-4b0c-896d-d2fb7111b640","resolution":{"observed_at":"2026-08-16T11:17:06.512196Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.17385","last_updated":"2025-04-17T17:59:27Z","snapshot_observed_at":"2026-08-16T13:06:51.567111Z","submitted_at":"2024-10-22T19:39:15Z","title":"Do Vision-Language Models Represent Space and How? Evaluating Spatial Frame of Reference Under Ambiguities","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.17385","snapshot_observed_at":"2026-08-16T05:29:57.046039Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.20648","last_updated":"2025-04-29T11:18:38Z","snapshot_observed_at":"2026-08-16T15:23:58.495936Z","submitted_at":"2025-04-29T11:18:38Z","title":"SpaRE: Enhancing Spatial Reasoning in Vision-Language Models with Synthetic Data","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-16T05:29:57.046039Z"},"links":{"cited_paper":"/paper/2410.17385","citing_paper":"/paper/2504.20648"},"observation_digest":"sha256:1174a6d9d225ea6728c100ae0a0f1e1c3dd613caa5c0ff083fb9618a295bae13","observation_id":"c6ecd8fb-e5d4-45e4-a9ae-ae54aa22204e","resolution":{"observed_at":"2026-08-16T05:29:57.046039Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.17385","last_updated":"2025-04-17T17:59:27Z","snapshot_observed_at":"2026-08-16T13:06:51.567111Z","submitted_at":"2024-10-22T19:39:15Z","title":"Do Vision-Language Models Represent Space and How? Evaluating Spatial Frame of Reference Under Ambiguities","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.17385","snapshot_observed_at":"2026-08-07T12:21:26.903958Z","title":"Do vision-language models represent space and how? evaluating spatial frame of reference under ambiguities","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.24870","last_updated":"2025-06-06T14:51:40Z","snapshot_observed_at":"2026-08-14T02:14:48.628979Z","submitted_at":"2025-05-30T17:59:26Z","title":"GenSpace: Benchmarking Spatially-Aware Image Generation","version":2},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-07T12:21:26.903958Z"},"links":{"cited_paper":"/paper/2410.17385","citing_paper":"/paper/2505.24870"},"observation_digest":"sha256:9d9573520e613b7ea1e987cc7ea8888aac9f5480dbc951fc9f86fcca24cf4d6d","observation_id":"c809b1ae-52e3-4d17-9928-ba1826c2861f","resolution":{"observed_at":"2026-08-07T12:21:26.903958Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.17385","last_updated":"2025-04-17T17:59:27Z","snapshot_observed_at":"2026-08-16T13:06:51.567111Z","submitted_at":"2024-10-22T19:39:15Z","title":"Do Vision-Language Models Represent Space and How? Evaluating Spatial Frame of Reference Under Ambiguities","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.17385","snapshot_observed_at":"2026-08-07T00:13:37.513161Z","title":"name \" :","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.14763","last_updated":"2025-06-17T17:57:37Z","snapshot_observed_at":"2026-08-16T07:38:15.131757Z","submitted_at":"2025-06-17T17:57:37Z","title":"RobotSmith: Generative Robotic Tool Design for Acquisition of Complex Manipulation Skills","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-07T00:13:37.513161Z"},"links":{"cited_paper":"/paper/2410.17385","citing_paper":"/paper/2506.14763"},"observation_digest":"sha256:7a81d61dca0a96d5a6690a0140e9b713dbcaafe6e8d9420bd89407ef14054630","observation_id":"5d8d4ce8-1ead-46aa-9a7a-4320a09cc668","resolution":{"observed_at":"2026-08-07T00:13:37.513161Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.17385","last_updated":"2025-04-17T17:59:27Z","snapshot_observed_at":"2026-08-16T13:06:51.567111Z","submitted_at":"2024-10-22T19:39:15Z","title":"Do Vision-Language Models Represent Space and How? Evaluating Spatial Frame of Reference Under Ambiguities","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.17385","snapshot_observed_at":"2026-08-15T18:28:06.277031Z","title":"K-level reasoning with large language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.19806","last_updated":"2026-07-12T11:46:09Z","snapshot_observed_at":"2026-08-16T07:16:21.862700Z","submitted_at":"2025-06-24T17:14:47Z","title":"LLM-Based Social Simulations Require a Boundary","version":3},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-15T18:28:06.277031Z"},"links":{"cited_paper":"/paper/2410.17385","citing_paper":"/paper/2506.19806"},"observation_digest":"sha256:8dcbba1c8147e1ff994f20e962cc178001d3e849ca6a140b7e7ccbe427b5c4fb","observation_id":"6cf7047d-dbdd-4062-b0e6-73561138ce58","resolution":{"observed_at":"2026-08-15T18:28:06.277031Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.17385","last_updated":"2025-04-17T17:59:27Z","snapshot_observed_at":"2026-08-16T13:06:51.567111Z","submitted_at":"2024-10-22T19:39:15Z","title":"Do Vision-Language Models Represent Space and How? Evaluating Spatial Frame of Reference Under Ambiguities","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.17385","snapshot_observed_at":"2026-08-06T14:33:44.971203Z","title":"org/CorpusID:272430309","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.22933","last_updated":"2025-07-24T16:27:38Z","snapshot_observed_at":"2026-08-15T15:23:34.215982Z","submitted_at":"2025-07-24T16:27:38Z","title":"Augmented Vision-Language Models: A Systematic Review","version":1},"reference_index":132,"source":"pdf_text","source_observed_at":"2026-08-06T14:33:44.971203Z"},"links":{"cited_paper":"/paper/2410.17385","citing_paper":"/paper/2507.22933"},"observation_digest":"sha256:e30780cb809887b1a0dc5c49393d794479b0714dca30a75350d59e6b422e6cd6","observation_id":"5d45d58b-8271-4058-967f-89ce26013993","resolution":{"observed_at":"2026-08-06T14:33:44.971203Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.17385","last_updated":"2025-04-17T17:59:27Z","snapshot_observed_at":"2026-08-16T13:06:51.567111Z","submitted_at":"2024-10-22T19:39:15Z","title":"Do Vision-Language Models Represent Space and How? Evaluating Spatial Frame of Reference Under Ambiguities","version":2},"cited_work":{"arxiv_id":"2410.17385","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2410.17385","snapshot_observed_at":"2026-07-02T21:07:24.216562Z","title":"Do vision-language models represent space and how? eval- uating spatial frame of reference under ambiguities","venue":null,"work_id":"f53c4559-e7da-40a9-935f-1e53b931b6be","year":2024},"citing_paper":{"arxiv_id":"2605.00273","last_updated":"2026-06-08T16:33:25Z","snapshot_observed_at":"2026-08-16T10:31:22.339382Z","submitted_at":"2026-04-30T22:18:33Z","title":"When Do Diffusion Models learn to Generate Multiple Objects?","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-07-01T08:07:10.345273Z"},"links":{"cited_paper":"/paper/2410.17385","citing_paper":"/paper/2605.00273"},"observation_digest":"sha256:f6d4e9e271819a5c0cfdb3e01ecac1e5b2b115d438b52d0570de50daa8eff1a4","observation_id":"c9abb6af-23de-4bc7-9c52-8f506d4a8604","resolution":{"observed_at":"2026-07-01T08:15:31.946430Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.17385","last_updated":"2025-04-17T17:59:27Z","snapshot_observed_at":"2026-08-16T13:06:51.567111Z","submitted_at":"2024-10-22T19:39:15Z","title":"Do Vision-Language Models Represent Space and How? Evaluating Spatial Frame of Reference Under Ambiguities","version":2},"cited_work":{"arxiv_id":"2410.17385","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2410.17385","snapshot_observed_at":"2026-07-02T21:07:24.216562Z","title":"Do vision-language models represent space and how? eval- uating spatial frame of reference under ambiguities","venue":null,"work_id":"f53c4559-e7da-40a9-935f-1e53b931b6be","year":2024},"citing_paper":{"arxiv_id":"2605.25524","last_updated":"2026-05-25T07:27:28Z","snapshot_observed_at":"2026-08-02T22:33:46.152327Z","submitted_at":"2026-05-25T07:27:28Z","title":"ProSR: Process-Shaped Spatial Reasoning for Reliable Chain-of-Thought in VLMs","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-29T22:23:38.178195Z"},"links":{"cited_paper":"/paper/2410.17385","citing_paper":"/paper/2605.25524"},"observation_digest":"sha256:4d73f40c57a6e6df25df370340d571d462cd0739baa9fcf711abde6262ad481f","observation_id":"64b0d813-e0cc-449e-95eb-d68b53d55d79","resolution":{"observed_at":"2026-06-29T22:23:59.865695Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.17385","last_updated":"2025-04-17T17:59:27Z","snapshot_observed_at":"2026-08-16T13:06:51.567111Z","submitted_at":"2024-10-22T19:39:15Z","title":"Do Vision-Language Models Represent Space and How? Evaluating Spatial Frame of Reference Under Ambiguities","version":2},"cited_work":{"arxiv_id":"2410.17385","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2410.17385","snapshot_observed_at":"2026-07-02T21:07:24.216562Z","title":"Do vision-language models represent space and how? eval- uating spatial frame of reference under ambiguities","venue":null,"work_id":"f53c4559-e7da-40a9-935f-1e53b931b6be","year":2024},"citing_paper":{"arxiv_id":"2606.08029","last_updated":"2026-06-06T07:45:19Z","snapshot_observed_at":"2026-08-12T17:40:58.190866Z","submitted_at":"2026-06-06T07:45:19Z","title":"IntentNav: Learning Spatial-Visual Object Navigation from Human Demonstrations","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-06-27T19:54:58.135809Z"},"links":{"cited_paper":"/paper/2410.17385","citing_paper":"/paper/2606.08029"},"observation_digest":"sha256:7f7f43e09cb4d804698d1f4ad9823b9fe03183b954bb60d5ff06d82bea64e673","observation_id":"91c09a56-d392-490c-a84f-47e490910923","resolution":{"observed_at":"2026-07-02T21:07:24.218032Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.17385","last_updated":"2025-04-17T17:59:27Z","snapshot_observed_at":"2026-08-16T13:06:51.567111Z","submitted_at":"2024-10-22T19:39:15Z","title":"Do Vision-Language Models Represent Space and How? Evaluating Spatial Frame of Reference Under Ambiguities","version":2},"cited_work":{"arxiv_id":"2410.17385","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2410.17385","snapshot_observed_at":"2026-07-02T21:07:24.216562Z","title":"Do vision-language models represent space and how? eval- uating spatial frame of reference under ambiguities","venue":null,"work_id":"f53c4559-e7da-40a9-935f-1e53b931b6be","year":2024},"citing_paper":{"arxiv_id":"2607.00881","last_updated":"2026-07-01T12:45:12Z","snapshot_observed_at":"2026-08-14T20:35:18.801176Z","submitted_at":"2026-07-01T12:45:12Z","title":"OmniView-Space: Reinforcing Spatial Reasoning via Multi-Perspective Spatial Mapping","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-07-02T14:16:39.649823Z"},"links":{"cited_paper":"/paper/2410.17385","citing_paper":"/paper/2607.00881"},"observation_digest":"sha256:545c816790e060a4455b3e47a9b6f0a32e17ad550bfa65d3265716de8251b99b","observation_id":"1e17cee6-cb6a-4eac-bc91-2b58da4e2a1a","resolution":{"observed_at":"2026-07-02T14:17:02.402742Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2410.17385/citation-record","integrity":"/paper/2410.17385/integrity","json":"/paper/2410.17385/citation-record.json","paper":"/paper/2410.17385"},"outbound":[],"paper":{"arxiv_id":"2410.17385","last_updated":"2025-04-17T17:59:27Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-16T13:06:51.567111Z","submitted_at":"2024-10-22T19:39:15Z","title":"Do Vision-Language Models Represent Space and How? Evaluating Spatial Frame of Reference Under Ambiguities"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"thesis":"As of 16 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 12 inbound Pith citation observations for arXiv:2410.17385."}