{"as_of":"2026-08-07T08:23:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:9fc2c0666fbd7f3ccd948c9461d9a6585d34d4d3ab9c76a9bcff3f34bd92a0fb","coverage":[{"denominator":58,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":58,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-11T00:56:06.243890Z","state":"measured"},{"denominator":59,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":59,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-01T06:23:00.372251Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-01T09:35:41.079576Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"cited_work":{"arxiv_id":"2605.07148","doi":null,"metadata_source":"pith","pith_arxiv_id":"2605.07148","snapshot_observed_at":"2026-07-01T09:35:41.079576Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","venue":"cs.CV","work_id":"4b0aa6ab-2eb7-410a-aac9-4014c46b1f4e","year":2026},"citing_paper":{"arxiv_id":"2606.31257","last_updated":"2026-06-30T07:33:18Z","snapshot_observed_at":"2026-07-07T00:04:58.486133Z","submitted_at":"2026-06-30T07:33:18Z","title":"Decodable Is Not Grounded: A Vision-Ablation Arbiter for VLM Spatial Reasoning","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-07-01T06:23:00.372251Z"},"links":{"cited_paper":"/paper/2605.07148","citing_paper":"/paper/2606.31257"},"observation_digest":"sha256:7fa3db9f181cf8e3f7cf73e7f33eac8a4df8a8d7f2bfc12bbc11babbe4f54ed5","observation_id":"b1f00412-5aeb-4dde-92e4-cf5cb283b46c","resolution":{"observed_at":"2026-07-01T09:35:41.081680Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2605.07148/citation-record","integrity":"/paper/2605.07148/integrity","json":"/paper/2605.07148/citation-record.json","paper":"/paper/2605.07148"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T12:44:53.794694Z","title":"Cognitive maps in rats and men.Psychological review, 55(4):189","venue":null,"work_id":"77df3af4-b729-4c39-8b2f-b60325a8372c","year":1948},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:4981150914497151c74175f187aa0188968144737aa705fa1e2180764c7a3bab","observation_id":"08c3477f-89d2-4975-8896-0434b2e46ef6","resolution":{"observed_at":"2026-05-16T00:51:59.835669Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Oxford university press","venue":null,"work_id":"69ec79ce-5dc2-4adb-9935-afee8b54b254","year":1978},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:be4191aa0ae8839b72cd7f65d647085e49f1ad0dde20e0baf39931443339ecc7","observation_id":"0f1b2dc9-aa01-48ea-92ff-d63faf01c576","resolution":{"observed_at":"2026-05-16T00:51:59.864222Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Non-euclidean navigation.Journal of Experimental Biology, 222(Suppl_1):jeb187971","venue":null,"work_id":"f39cbf26-95d1-4852-bcd6-27eb556bd52c","year":2019},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:4ed031168794aad0d8d7f1ae0f7e1f04e3bafc90f5868c6da76eee3b7f973c7a","observation_id":"f6200967-202b-452d-8c8f-86f4c8d91f81","resolution":{"observed_at":"2026-05-16T00:51:59.776402Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Structuring knowledge with cognitive maps and cognitive graphs.Trends in cognitive sciences, 25(1):37–54","venue":null,"work_id":"2574aad4-9536-4f3f-9866-57f4d2e43036","year":2021},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:66073cb61aef19bc68ca25e5d2d11ab49d8d67d8444934ac3da57692d710a441","observation_id":"cd19fa2b-720d-4c83-96cd-69a6048d47bc","resolution":{"observed_at":"2026-05-16T00:51:59.750044Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03326","last_updated":"2024-10-26T16:35:13Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-06T17:59:44Z","title":"LLaVA-OneVision: Easy Visual Task Transfer","version":3},"cited_work":{"arxiv_id":"2408.03326","doi":"10.48550/arxiv.2408.03326","metadata_source":"pith","pith_arxiv_id":"2408.03326","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LLaVA-OneVision: Easy Visual Task Transfer","venue":"cs.CV","work_id":"f5f2452b-f2a9-49ac-b38d-c76e18cdfe49","year":2024},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"cited_paper":"/paper/2408.03326","citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:4b611a6fcd20855503c936b555a25aa9ad595fa6a6b0f861e4a96d98b582484c","observation_id":"27281b9e-500c-480a-bf0e-caf4a31cd9d7","resolution":{"observed_at":"2026-05-11T05:00:55.062047Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Video-chatgpt: Towards detailed video understanding via large vision and language models","venue":null,"work_id":"50c5e35b-7241-4f6b-835b-23a10d4e3583","year":2024},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:20f444d083565b9dae9a57e63271635c60946a8162cdb119d223f8ec43a4bb10","observation_id":"5155a600-edc0-47e2-8798-bfc8cfff8a9d","resolution":{"observed_at":"2026-05-16T00:51:59.830776Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.13246","last_updated":"2024-10-10T22:22:52Z","snapshot_observed_at":"2026-07-06T18:33:28.511600Z","submitted_at":"2024-06-19T06:15:26Z","title":"GSR-BENCH: A Benchmark for Grounded Spatial Reasoning Evaluation via Multimodal LLMs","version":2},"cited_work":{"arxiv_id":"2406.13246","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.13246","snapshot_observed_at":"2026-06-29T07:53:13.566074Z","title":"Gsr-bench: A benchmark for grounded spatial reasoning evaluation via multimodal llms","venue":null,"work_id":"e54859c5-3a14-4b51-b4a1-25ddf6c0e3d7","year":2024},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"cited_paper":"/paper/2406.13246","citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:8f54a96f8004914021552db43ccac3332b2858b7ec5f095acfe1eb5d233520a4","observation_id":"ee7d748a-dd6c-4c90-a4d3-f1fe73c1ad7f","resolution":{"observed_at":"2026-05-11T05:00:55.057963Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.07474","last_updated":"2023-04-12T20:05:41Z","snapshot_observed_at":"2026-07-06T14:05:02.813765Z","submitted_at":"2022-10-14T02:52:26Z","title":"SQA3D: Situated Question Answering in 3D Scenes","version":5},"cited_work":{"arxiv_id":"2210.07474","doi":null,"metadata_source":"pith","pith_arxiv_id":"2210.07474","snapshot_observed_at":"2026-07-11T01:37:43.684631Z","title":"Sqa3d: Situated question answering in 3d scenes","venue":"cs.CV","work_id":"8d9797f2-5882-4b0e-99d7-62e075c67387","year":2022},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"cited_paper":"/paper/2210.07474","citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:634801f594aaf99b9dbf8fb6baf273174f5d4bc520b4efbdc6704b91ba898b7a","observation_id":"ef9c6ca6-02f4-4ae1-ab5f-df39ffd73df5","resolution":{"observed_at":"2026-05-11T05:00:55.089512Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Reasoning paths with reference objects elicit quantitative spatial reasoning in large vision-language models","venue":null,"work_id":"dd178eb0-5904-4b28-a4c1-fa0418070b7a","year":2024},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:f77543d9f98de162c4b848e206684636ea85a5fde60fe14d584094a009b6b8d6","observation_id":"8e743d46-bcb1-403b-86ee-2961d9043cd8","resolution":{"observed_at":"2026-05-16T00:51:59.787266Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2511.15722","doi":"10.48550/arxiv.2511.15722","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Spatial reasoning in multimodal large language models: A survey of tasks, benchmarks and methods","venue":"ArXiv.org","work_id":"6a89950e-e68f-4ee1-92ca-4c617487ec7c","year":2025},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:5d7ab50b1893aa31415b5645ef01bddf09d7d9bfb8b34a545186cab9fcbba12e","observation_id":"927ebeab-a5bc-4728-99de-eb2cbe62b1dc","resolution":{"observed_at":"2026-05-11T05:00:55.110438Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2511.18200","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Infinibench: Infinite benchmarking for visual spatial reasoning with customizable scene complexity","venue":null,"work_id":"90525787-b2d8-455f-b7d2-f962c4a35167","year":2025},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:00e99a97be7e352671adfb4730f207e305f7ea73751b0550d85e1a19f4ed9036","observation_id":"6b8a87e7-0520-4788-87c1-62fa8f53e181","resolution":{"observed_at":"2026-05-11T05:00:55.177210Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Generic attention-model explainability for interpreting bi-modal and encoder-decoder transformers","venue":null,"work_id":"ca2e0f67-a0c5-4e40-b440-6d840941b33e","year":2021},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:a9dc619eb874a511a19756a03586e9154299eea89d3404e031f30f5b39cf1865","observation_id":"8e86367b-a397-4a40-bd3a-17dffc49096b","resolution":{"observed_at":"2026-05-16T00:51:59.855459Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2503.01773","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T06:39:37.674682Z","title":"Why is spatial reasoning hard for vlms? an attention mechanism perspective on focus areas","venue":null,"work_id":"cb9bb652-e52c-4ea5-9f44-b1c3342e7c5c","year":2025},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:e2191bc453875e7e586168bbb89597fa5a7953798f2f947079994abeda2d1fdb","observation_id":"85e2888c-989e-427a-b271-75f040e6cd36","resolution":{"observed_at":"2026-05-11T05:00:55.093786Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2503.17349","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T00:27:30.257312Z","title":"Beyond semantics: Rediscovering spatial awareness in vision-language models","venue":null,"work_id":"c661cee3-78d3-4e40-8755-e8221cb6aac7","year":2025},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:ef8484b37c6029be14329b0ad9f9f4fe5d625979b6f4c37e67557d8c2194b060","observation_id":"614fd597-f426-4038-920b-02175952746f","resolution":{"observed_at":"2026-05-11T05:00:55.140796Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01506","last_updated":"2025-02-18T02:23:45Z","snapshot_observed_at":"2026-08-07T07:48:23.574313Z","submitted_at":"2024-06-03T16:34:01Z","title":"The Geometry of Categorical and Hierarchical Concepts in Large Language Models","version":3},"cited_work":{"arxiv_id":"2406.01506","doi":"10.48550/arxiv.2406.01506","metadata_source":"pith","pith_arxiv_id":"2406.01506","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2406.01506 , year=","venue":"cs.CL","work_id":"50fe71d6-8df7-4aba-81c3-41f35bcc40c7","year":2024},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"cited_paper":"/paper/2406.01506","citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:6e850e4ed61fad2cc0e96aa78385ab60dd02e4451a7d8706f426941ae632eda2","observation_id":"c2cc8c22-7df3-492c-83e6-ca6d44185f43","resolution":{"observed_at":"2026-05-11T05:00:54.996003Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2601.12626","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-01T09:35:41.058794Z","title":"Linear mechanisms for spa- tiotemporal reasoning in vision language models.arXiv preprint arXiv:2601.12626","venue":null,"work_id":"90de4c0a-88ef-4a8f-b2fc-4a94864f4949","year":2026},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:53899d7847f0166fb149b7f0787d87942a34dfa9904fddf0091cac1517bf5254","observation_id":"c6a61006-fe26-4a2a-82bb-a79f507e37c3","resolution":{"observed_at":"2026-05-11T05:00:55.078944Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2506.15871","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Visual symbolic mechanisms: Emergent sym- bol processing in vision language models","venue":null,"work_id":"095aef16-3a07-474d-87fe-44a3e7b1bb6e","year":2025},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:b89e2153f3b4126bb5d040b6ad597865b3ad08d29e8b4bec160d7e14a44b3f3c","observation_id":"5d1eb770-c5a5-4726-b3d8-4ce8a463a431","resolution":{"observed_at":"2026-05-11T05:00:55.074503Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Analyzing the behavior of visual question answering models","venue":null,"work_id":"23692b35-5b8a-4ed3-8494-347f9ac4e922","year":2016},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:de42ad6dcdc2ca0b1634e74c5f92cf47170c54192ba3d6a402211bf27aae9fb3","observation_id":"49f331ee-9b96-4d8e-a761-040cd6933770","resolution":{"observed_at":"2026-05-16T00:51:59.821555Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T15:02:40.786929Z","title":"Making the v in vqa matter: Elevating the role of image understanding in visual question answering","venue":null,"work_id":"74f5a87a-5a29-4215-961b-853783e57bbb","year":2017},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:9c3d07b8c8068a1add063f43d034aa1e15718f69478793e8b04552abdd83948e","observation_id":"9bae1465-c803-42dc-adac-5b739f011eab","resolution":{"observed_at":"2026-05-16T00:51:59.797458Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2602.22394","last_updated":"2026-04-14T12:53:45Z","snapshot_observed_at":"2026-07-06T22:47:06.893212Z","submitted_at":"2026-02-25T20:42:35Z","title":"Vision Transformers Need More Than Registers","version":2},"cited_work":{"arxiv_id":"2602.22394","doi":null,"metadata_source":"pith","pith_arxiv_id":"2602.22394","snapshot_observed_at":"2026-06-30T23:15:07.982818Z","title":"Vision Transformers Need More Than Registers","venue":"cs.CV","work_id":"3cc1a4cc-ea92-4c36-a634-6a82be06a9d6","year":2026},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"cited_paper":"/paper/2602.22394","citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:15ff20f4f4e3145fd6f66ec43b73cef25508b11f85c639ad3bfa6d4f301e79eb","observation_id":"037a3f59-e3e9-40d4-9a48-0538a26a2384","resolution":{"observed_at":"2026-05-11T05:00:55.183613Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T00:31:42.304430Z","title":"Thinking in space: How multimodal large language models see, remember, and recall spaces","venue":null,"work_id":"c6a4e059-9ea2-4efc-9e59-970019869797","year":2025},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:3f707279380bc669af5fb27cf77871e3125e1fc9883f84345e064f9b5f53c5f7","observation_id":"5b3b7ce2-3770-4241-add7-677b6fecd296","resolution":{"observed_at":"2026-05-16T00:51:59.817349Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Mindcube: Spatial mental modeling from limited views.arXiv e-prints, pages arXiv–2506","venue":null,"work_id":"804136f4-8290-42df-b0a7-48d2dfcca14b","year":2025},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:285a8ed7e75c484dd56b11ad940a1d332b44cb7444b7a58179bbfc0f6aeb0c14","observation_id":"a3c8bfe4-cf8e-4839-a798-c2f68178f098","resolution":{"observed_at":"2026-05-16T00:51:59.840123Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2602.07082","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T17:49:59.941555Z","title":"arXiv preprint arXiv:2602.07082 (2026),https://arxiv.org/ abs/2602.070824","venue":null,"work_id":"1ffb320c-b376-400e-b024-644f72dc2db4","year":2026},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:e9adb9d82d62333ae5e0a0555e5d9a82a680dfdc4aad61dab2036ec03a8f4ce9","observation_id":"87938ae6-6ca2-4747-b61f-b550c280af50","resolution":{"observed_at":"2026-05-11T05:00:55.121495Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T00:31:42.271336Z","title":"Spatialrgpt: Grounded spatial reasoning in vision-language models","venue":null,"work_id":"937b3c04-07e3-44e7-8d9e-eacfd62869c6","year":2024},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:f35f4077809336c159276ca2b7a28288a5a118128760ede2d971ad239384e2ee","observation_id":"181f7c00-1c9a-4297-9606-1a175e9b9ef2","resolution":{"observed_at":"2026-05-16T00:51:59.850247Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2511.13719","doi":"10.48550/arxiv.2511.13719","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Scaling spatial intelligence with multimodal foundation models","venue":"arXiv (Cornell University)","work_id":"ceb84861-8bdb-4822-864c-8e2d0ae4d322","year":2025},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:f1d4bebba42fb18ae827b5bc598756cedb404b2ff12dc2edbb477094284bc856","observation_id":"338621f3-c352-40ec-801e-fa934f5f5090","resolution":{"observed_at":"2026-05-11T05:00:55.013102Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2511.07403","last_updated":"2026-07-02T19:21:24Z","snapshot_observed_at":"2026-08-07T07:35:15.256090Z","submitted_at":"2025-11-10T18:52:47Z","title":"SpatialThinker: Reinforcing Scene Graph-Grounded Spatial Reasoning via Dense Rewards","version":2},"cited_work":{"arxiv_id":"2511.07403","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2511.07403","snapshot_observed_at":"2026-07-07T01:16:00.407423Z","title":"Spatialthinker: Reinforcing 3d reasoning in multimodal llms via spatial rewards","venue":null,"work_id":"1d926991-cdce-49ae-9ba1-226aca6c7ec8","year":2025},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"cited_paper":"/paper/2511.07403","citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:439d9ea7c519c021237685cdc42cf74ccd0e18d06169c1a5bbafd19a6816eecb","observation_id":"ca4585f3-9bbd-455f-99f5-b88ee5681fbf","resolution":{"observed_at":"2026-07-07T01:16:00.407423Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.09965","last_updated":"2025-06-19T03:46:55Z","snapshot_observed_at":"2026-07-31T21:40:49.363128Z","submitted_at":"2025-06-11T17:41:50Z","title":"Reinforcing Spatial Reasoning in Vision-Language Models with Interwoven Thinking and Visual Drawing","version":2},"cited_work":{"arxiv_id":"2506.09965","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.09965","snapshot_observed_at":"2026-07-05T11:41:02.594727Z","title":"Reinforcing Spatial Reasoning in Vision-Language Models with Interwoven Thinking and Visual Drawing","venue":"cs.CV","work_id":"ff3bffaa-30ad-4319-b157-10557ee89344","year":2025},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"cited_paper":"/paper/2506.09965","citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:2b2c82a93fd0dcd33986bddfd72483e26ae7c100a6ba1488aa218400e2ab3142","observation_id":"2d33efec-352e-454e-86c7-bb00797f260e","resolution":{"observed_at":"2026-05-17T04:58:10.438799Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Causal abstractions of neural networks.Advances in neural information processing systems, 34:9574–9586","venue":null,"work_id":"bdff419a-6d02-4648-8167-ff84746236b8","year":2021},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:943c91dccccaebae2a9188fee54d609cd5d7a298a76dfd6b55b9a3d45c145184","observation_id":"37da16b4-2c4e-4f13-9efc-754e4fa9f5c7","resolution":{"observed_at":"2026-05-16T00:51:59.812012Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-10T15:47:23.611428Z","title":"Investigating gender bias in language models using causal mediation analysis.Advances in neural information processing systems, 33:12388–12401","venue":null,"work_id":"a0f81014-6d97-4e06-a958-227d6fed181a","year":2020},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:38eb5ccca6b079bb26fc016751ef62fdbaaffa5f84f17f28467fd04cdee68f31","observation_id":"b410ba3a-7c41-4aed-b402-320a88e5580e","resolution":{"observed_at":"2026-05-16T00:51:59.771191Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Emergent linear representations in world models of self-supervised sequence models","venue":null,"work_id":"6ce541af-541b-4b89-bcd1-1c7d0a202dc8","year":2023},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:2b3787e87a0f51c3814dbcd2e8c86a85e5a90c9cb127b7100b6c7d71a8ae66c0","observation_id":"b743b19c-20f2-4c89-b492-e9dba5bb284c","resolution":{"observed_at":"2026-05-16T00:51:59.760153Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.01916","last_updated":"2026-05-14T06:57:50Z","snapshot_observed_at":"2026-08-05T18:32:08.286538Z","submitted_at":"2025-08-03T20:59:29Z","title":"Decomposing Representation Space into Interpretable Subspaces with Unsupervised Learning","version":3},"cited_work":{"arxiv_id":"2508.01916","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2508.01916","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Decomposing representation space into interpretable subspaces with unsupervised learning","venue":null,"work_id":"7bcb55cd-741e-49bf-a03a-e63ef918a4fe","year":2025},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"cited_paper":"/paper/2508.01916","citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:e71587ec3626e9e6c8a14b68e6fed5ed4c3b7a52b212c9edddd24d241c83fb16","observation_id":"cfd72f98-76e0-4803-b261-18f66a79e08b","resolution":{"observed_at":"2026-05-15T01:43:10.747895Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.24709","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T00:39:16.389687Z","title":"Does object binding naturally emerge in large pretrained vi- sion transformers?","venue":null,"work_id":"438d845f-61a9-464c-8ac2-0869817a499b","year":2026},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:f69db5bb50123495220f714eec638fd377aac00368954a6f7ee0f16679b065d9","observation_id":"0c5c4716-1a45-4ccb-b2ce-10b46640e1ac","resolution":{"observed_at":"2026-05-11T05:00:55.021812Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2511.21471","last_updated":"2026-05-07T07:59:46Z","snapshot_observed_at":"2026-08-02T23:27:20.204280Z","submitted_at":"2025-11-26T15:04:18Z","title":"SpatialBench: Benchmarking Multimodal Large Language Models for Spatial Cognition","version":4},"cited_work":{"arxiv_id":"2511.21471","doi":null,"metadata_source":"pith","pith_arxiv_id":"2511.21471","snapshot_observed_at":"2026-07-04T08:59:43.334805Z","title":"SpatialBench: Benchmarking Multimodal Large Language Models for Spatial Cognition","venue":"cs.AI","work_id":"81942f57-c90d-41fc-9036-779f35d52bed","year":2025},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"cited_paper":"/paper/2511.21471","citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:0faff8f652b3a2251981061ed9603a7cafaab7e8a292005199cf6357b4e15e23","observation_id":"cbe57a86-3ae8-4a3f-a2d5-0c3ca7acd9a5","resolution":{"observed_at":"2026-05-11T05:00:55.152020Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2509.09676","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T07:56:04.737468Z","title":"Spatialvid: A large-scale video dataset with spatial annotations.arXiv preprint arXiv:2509.09676, 2025a","venue":null,"work_id":"fa42660a-b999-4c7f-80c9-2b72151c3399","year":2025},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:107aa00ad1329fa1510ef77bb821f026966ef5e3a584234de930809332a75434","observation_id":"058cd24e-667c-4fd4-a396-ed5a7c8120bb","resolution":{"observed_at":"2026-05-11T05:00:55.048555Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Qwen2.5-vl, January 2025","venue":null,"work_id":"5ddd69c1-ac73-4146-ab25-6b0a8d53ae9a","year":2025},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:2d6fd12e49d7bbbd295e22ff7e661b6abbae7e6365afdfcd419fff891d4fcd4a","observation_id":"4ef4ad3a-7120-4fda-a182-e177201cabc0","resolution":{"observed_at":"2026-05-16T00:51:59.803547Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.10479","last_updated":"2025-04-19T03:47:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-14T17:59:25Z","title":"InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models","version":3},"cited_work":{"arxiv_id":"2504.10479","doi":"10.48550/arxiv.2504.10479","metadata_source":"pith","pith_arxiv_id":"2504.10479","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models","venue":"cs.CV","work_id":"fe8637aa-12bc-4434-8d36-9f57b5eebcbe","year":2025},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"cited_paper":"/paper/2504.10479","citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:dd1761d46ba83ca8b0c2f1fd60c235ce334bbce1f5babd375a3e68c1363417c0","observation_id":"0bbd56e6-483b-4a62-8228-81bf73abc4eb","resolution":{"observed_at":"2026-05-11T05:00:55.145785Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-05-20T07:54:09.017512+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T07:54:09.017512+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.15154","last_updated":"2023-10-23T17:55:31Z","snapshot_observed_at":"2026-07-06T16:37:19.963994Z","submitted_at":"2023-10-23T17:55:31Z","title":"Linear Representations of Sentiment in Large Language Models","version":1},"cited_work":{"arxiv_id":"2310.15154","doi":"10.48550/arxiv.2310.15154","metadata_source":"pith","pith_arxiv_id":"2310.15154","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Linear Representations of Sentiment in Large Language Models","venue":"cs.LG","work_id":"6cb3c7a7-3301-449f-97b9-7e047edafdf9","year":2023},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"cited_paper":"/paper/2310.15154","citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:10b104a857bf4f3f6b6161e53703c7a724008068b66a91e67b3d9524368ed3ed","observation_id":"6493fa68-41b8-4a0b-a46c-d4e7f916c00e","resolution":{"observed_at":"2026-05-15T12:43:06.705954Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Linear spaces of meanings: compositional structures in vision-language models","venue":null,"work_id":"e9f5f39a-16a3-477d-8236-6a1b80599e19","year":2023},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:8225b36d6dcb215bc8dd20fb960df8151897e6e754af838ecbe5c4f0379a333b","observation_id":"7813ad13-0ad4-425f-bb5b-dd6db8a09fc0","resolution":{"observed_at":"2026-05-16T00:51:59.825709Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2511.01932","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Deciphering personalization: Towards fine-grained explainability in natural language for personalized image generation models","venue":null,"work_id":"80887d6c-746d-4012-a28b-9aa0435f0252","year":2025},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:327e513dfec76451d45562643c51c4a246328a8236cc0727f40b488802a586b4","observation_id":"c4936031-72cf-45e6-bedf-49071b82a667","resolution":{"observed_at":"2026-05-11T05:00:55.156748Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.03658","last_updated":"2024-07-17T22:24:27Z","snapshot_observed_at":"2026-07-06T16:43:58.947915Z","submitted_at":"2023-11-07T01:59:11Z","title":"The Linear Representation Hypothesis and the Geometry of Large Language Models","version":2},"cited_work":{"arxiv_id":"2311.03658","doi":"10.48550/arxiv.2311.03658","metadata_source":"pith","pith_arxiv_id":"2311.03658","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"The Linear Representation Hypothesis and the Geometry of Large Language Models","venue":"cs.CL","work_id":"a7b44adc-f2c2-4420-a27d-8ade97dd3b75","year":2023},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"cited_paper":"/paper/2311.03658","citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:37d9f7f7655a7ca42c955aa6b8c08395a1d63918e4d2315854d72f72035ce013","observation_id":"910488ed-2c93-4e9a-8e62-f61b379587e9","resolution":{"observed_at":"2026-05-11T21:43:32.339354Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-18T08:21:05.185759+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-18T08:21:05.185759+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.03867","last_updated":"2024-03-06T17:17:36Z","snapshot_observed_at":"2026-08-04T18:22:42.733859Z","submitted_at":"2024-03-06T17:17:36Z","title":"On the Origins of Linear Representations in Large Language Models","version":1},"cited_work":{"arxiv_id":"2403.03867","doi":"10.48550/arxiv.2403.03867","metadata_source":"arxiv_reference","pith_arxiv_id":"2403.03867","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"2024 , month = mar, number =","venue":"arXiv (Cornell University)","work_id":"a8373be4-a6ad-40d5-8432-45f186aa5ad4","year":2024},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"cited_paper":"/paper/2403.03867","citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:8d7e02c8456937c0973e8543166018dc26fc5dbc4b20aa0c83d57add97245a8e","observation_id":"9fe6844e-8149-443b-8446-9f62bf4d1938","resolution":{"observed_at":"2026-05-11T05:00:55.069345Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.04706","last_updated":"2025-06-05T07:30:58Z","snapshot_observed_at":"2026-08-05T23:42:32.012691Z","submitted_at":"2025-06-05T07:30:58Z","title":"Line of Sight: On Linear Representations in VLLMs","version":1},"cited_work":{"arxiv_id":"2506.04706","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.04706","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Line of sight: On linear representations in vllms","venue":null,"work_id":"120ccc7d-1a6b-4b4f-8e05-cd620c47078c","year":2025},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"cited_paper":"/paper/2506.04706","citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:9a5a0c4f9c97835a67ed6b2304ae6ddcbd0b2bb63d2bf13bfc0d6815d42eac2a","observation_id":"cad0295a-cb29-49e2-b2e8-09af52fc21fc","resolution":{"observed_at":"2026-05-11T05:00:55.043918Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Interpreting clip with sparse linear concept embeddings (splice).Advances in Neural Information Processing Systems, 37:84298–84328","venue":null,"work_id":"6f30fba9-13c9-421a-bd87-830550d66c4a","year":2024},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:bab0b1fc580ad63f1c3941eba4cd8606907139ed6750f40f46f5c62622c5fc12","observation_id":"203ca3fe-6312-480a-8f52-b33d225eff26","resolution":{"observed_at":"2026-05-16T00:51:59.766403Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.02996","last_updated":"2025-06-03T15:31:00Z","snapshot_observed_at":"2026-08-03T16:16:16.239767Z","submitted_at":"2025-06-03T15:31:00Z","title":"Linear Spatial World Models Emerge in Large Language Models","version":1},"cited_work":{"arxiv_id":"2506.02996","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.02996","snapshot_observed_at":"2026-07-01T09:35:41.083419Z","title":"arXiv preprint arXiv:2506.02996 , year=","venue":null,"work_id":"5935146b-683d-4432-a1f7-ec6cd953bc8a","year":2025},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"cited_paper":"/paper/2506.02996","citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:da4d4de3f059fd7cf2084a5fbfee4c13cb90059769af0af3d06121ba5001a15f","observation_id":"0d0fdd4e-6d1a-4a93-9455-6b94734e17c1","resolution":{"observed_at":"2026-05-11T05:00:54.983163Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Laplacian eigenmaps for dimensionality reduction and data representation.Neural computation, 15(6):1373–1396","venue":null,"work_id":"ed1bd55d-4e5b-400a-8a7e-890c11077e4e","year":2003},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:e12eeb0813d57753d4a9ff019c579ab2331cdaf4e3e0c4d10cbb0b27175a9251","observation_id":"fc1baed4-e6aa-4312-a630-77e07bc6c1e4","resolution":{"observed_at":"2026-05-16T00:51:59.742602Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.00070","last_updated":"2025-05-02T05:27:38Z","snapshot_observed_at":"2026-08-04T09:15:50.354490Z","submitted_at":"2024-12-29T18:58:09Z","title":"ICLR: In-Context Learning of Representations","version":2},"cited_work":{"arxiv_id":"2501.00070","doi":"10.48550/arxiv.2501.00070","metadata_source":"arxiv_reference","pith_arxiv_id":"2501.00070","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Iclr: In-context learning of representations","venue":"arXiv (Cornell University)","work_id":"9ccab950-5e6c-40c5-90fb-4df391ad5cf3","year":2024},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"cited_paper":"/paper/2501.00070","citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:0f049a205fb1eb2d1517bcea2ef5960343292f7a75feab0c742d50173331c9de","observation_id":"689aa3bb-4778-4f13-9d7f-a1543783fed9","resolution":{"observed_at":"2026-05-11T05:00:55.026979Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.21040","last_updated":"2025-07-28T17:56:34Z","snapshot_observed_at":"2026-08-07T05:01:02.602197Z","submitted_at":"2025-07-28T17:56:34Z","title":"Transformers as Unrolled Inference in Probabilistic Laplacian Eigenmaps: An Interpretation and Potential Improvements","version":1},"cited_work":{"arxiv_id":"2507.21040","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.21040","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Transformers as unrolled inference in probabilis- tic laplacian eigenmaps: An interpretation and potential improvements","venue":null,"work_id":"66a0c690-a4f0-4963-880e-0e7993f321d7","year":2025},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"cited_paper":"/paper/2507.21040","citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:49644e5d7b3de0168516e5d697629b87a5b38705cbca03efc722844743ae1cfc","observation_id":"cabe0242-351e-46ee-8f59-8b5a6bfc806d","resolution":{"observed_at":"2026-05-11T05:00:55.098204Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.06811","last_updated":"2024-10-27T23:45:38Z","snapshot_observed_at":"2026-07-06T18:28:33.795339Z","submitted_at":"2024-06-10T21:34:43Z","title":"Learning Continually by Spectral Regularization","version":2},"cited_work":{"arxiv_id":"2406.06811","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.06811","snapshot_observed_at":"2026-07-04T19:30:06.975725Z","title":"arXiv preprint arXiv:2406.06811 , year=","venue":null,"work_id":"9215f5b9-e4a1-4dfb-9a8f-f10665139ee5","year":2024},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"cited_paper":"/paper/2406.06811","citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:d9f6f9fed7d3d33365f70bd020093a5a6529591a8f94e98b9354d252c6e125c9","observation_id":"5ec14619-68aa-49f3-8e0b-34c367b7f2d9","resolution":{"observed_at":"2026-05-11T05:00:54.975635Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Principal spectral regularization makes momentum surpass adam for llm training","venue":null,"work_id":"5a9a808c-019f-4395-906e-5d6e1c0ab246","year":null},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:92ca843bcc1260cfc05f96bf3e754b25edea5d3d54338351b6a38c70bdc69af8","observation_id":"6d13ab07-fc8f-4fbc-8322-3c90687f59e2","resolution":{"observed_at":"2026-05-16T00:51:59.860201Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1709.05584","last_updated":"2018-04-10T15:26:32Z","snapshot_observed_at":"2026-07-06T05:59:59.763065Z","submitted_at":"2017-09-17T00:19:33Z","title":"Representation Learning on Graphs: Methods and Applications","version":3},"cited_work":{"arxiv_id":"1709.05584","doi":null,"metadata_source":"pith","pith_arxiv_id":"1709.05584","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Representation Learning on Graphs: Methods and Applications","venue":"cs.SI","work_id":"429ab37e-5fbc-4365-bef5-846397505f60","year":2017},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"cited_paper":"/paper/1709.05584","citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:42dfb1fb0a2658834c020cc33f5bb3227279dcbee5456d5768d7174f6f10978c","observation_id":"337f0e07-c135-47fd-a09f-8002e729bb5c","resolution":{"observed_at":"2026-05-11T05:00:55.162225Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1609.02907","last_updated":"2017-02-22T09:55:36Z","snapshot_observed_at":"2026-07-06T05:10:16.862707Z","submitted_at":"2016-09-09T19:48:41Z","title":"Semi-Supervised Classification with Graph Convolutional Networks","version":4},"cited_work":{"arxiv_id":"1609.02907","doi":"10.1039/d0cp00305k","metadata_source":"pith","pith_arxiv_id":"1609.02907","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Semi-Supervised Classification with Graph Convolutional Networks","venue":"cs.LG","work_id":"21fff118-807d-49cd-8229-f7087ba57b5d","year":2016},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"cited_paper":"/paper/1609.02907","citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:2fc2ffcc880e44a5f87ad5d1d2fb710b3c7c6df914383ddf82d1e1791fba606e","observation_id":"94272ea8-8158-4e44-879e-c6267b7de011","resolution":{"observed_at":"2026-05-11T05:00:55.187460Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"cognitive-map","venue":null,"work_id":"b87e64ca-608b-4982-9546-aeeb7951d4cd","year":2003},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:7053dc6a5dce1fa7fbf9816efb4e7fe46a0d1d37eca2c1ab3ddc2d7f0753cd8f","observation_id":"2d70da38-6d48-4dc0-9a45-ffe457edcad4","resolution":{"observed_at":"2026-05-16T00:51:59.735211Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"d76a7452-2e92-4280-8b67-6966065dff87","year":null},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:766b9242be5c09f7fc5daa5289a49cc7ac5e669bad337c9492499ac7f6747418","observation_id":"5c985aba-b092-4bf8-8ed1-a8b7c7065b12","resolution":{"observed_at":"2026-05-16T00:51:59.868005Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"52b21c58-f9a6-427c-97bb-5be93f51588b","year":null},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:5b75003bf0eb095080b0a589f93a3ee838be7ad51054ff248fc47456d7c93719","observation_id":"5b09b197-e4c4-48ec-a358-ca4d05ba1cbe","resolution":{"observed_at":"2026-05-16T00:51:59.807668Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Each row is rescaled to unit norm","venue":null,"work_id":"76913e54-a6ef-47b1-86d8-106c2c796923","year":null},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:55cf54f707d65ff48efc4ce0e0aa2d97c2182de64547b5f73bd31cc22cecc151","observation_id":"65e336db-7681-45b7-82c0-613d46b187b9","resolution":{"observed_at":"2026-05-16T00:51:59.781489Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Drop columns with diagonal |Rkk| below 10−6 of the maximum (rank-deficient class directions)","venue":null,"work_id":"aaaf188e-9628-4ec7-ba35-7ead716a248b","year":null},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:281241d830a63864de3f78ebc57d99e70786fefb1b7aaa2cf464a03c6e5b9420","observation_id":"c8697e43-b4f4-4bc0-8f19-1de257aafae3","resolution":{"observed_at":"2026-05-16T00:51:59.791989Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Algorithm summary.The end-to-end per-step computation is: 1.Forward pass, capturingH ℓ ∈R B×T×d at each Dirichlet-target layer via forward hooks","venue":null,"work_id":"29921fda-f449-434e-b18e-2a510a9d0231","year":null},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:7179d74d0f77316d55c550e68d8faf03c88ac9a9971d335a8cf298780eda49d3","observation_id":"512bef4f-93ca-4709-b63c-60a3c64a58b3","resolution":{"observed_at":"2026-05-16T00:51:59.754920Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"For the residMulti variant, steps 1–8 run independently at 5 layers and step 9 averages the resulting ratios","venue":null,"work_id":"053638b6-d526-4ebb-976e-7250ca418e51","year":null},"citing_paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-05-11T00:56:06.243890Z"},"links":{"citing_paper":"/paper/2605.07148"},"observation_digest":"sha256:3575d562e54466bc38b3459ee3168e04d15ab5defda722b0df0dc889f49fb8dd","observation_id":"512158b3-f242-451b-a783-c838eca33480","resolution":{"observed_at":"2026-05-16T00:51:59.844562Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2605.07148","last_updated":"2026-05-08T02:32:27Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T23:19:30.387067Z","submitted_at":"2026-05-08T02:32:27Z","title":"Uncovering and Shaping the Latent Representation of 3D Scene Topology in Vision-Language Models"},"reference_resolution":{"displayed":58,"state_counts":{"malformed_identifier":0,"metadata_mismatch":3,"parse_uncertain":0,"unresolved":2,"verified_exact":28,"verified_fuzzy":25},"total_outbound_references":58},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 58 of 58 outbound references and 1 inbound Pith citation observation for arXiv:2605.07148."}