{"as_of":"2026-08-06T01:51:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:3f7dffbd715c65ee99cbb4f979f1a05d996f5be44049dfbc452f2669fa8cd000","coverage":[{"denominator":94,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":94,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-27T16:35:14.099586Z","state":"measured"},{"denominator":95,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":95,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-05T06:32:48.257954+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-31T05:01:26.556632Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2606.09669","snapshot_observed_at":"2026-07-31T05:01:26.556632Z","title":"arXiv preprint arXiv:2606.09669 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.24957","last_updated":"2026-07-27T18:04:54Z","snapshot_observed_at":"2026-08-01T22:57:25.487048Z","submitted_at":"2026-07-27T18:04:54Z","title":"PerceptionBench: Evaluating Atomic Visual Perception in Multimodal Large Language Models","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-07-31T05:01:26.556632Z"},"links":{"cited_paper":"/paper/2606.09669","citing_paper":"/paper/2607.24957"},"observation_digest":"sha256:a6d5318debc39d0eae6de8eeee9c7851c633f56fddf4dd6e0fbc94ec6aa0f323","observation_id":"38384984-b82d-4c47-a2c8-d907d08dc60d","resolution":{"observed_at":"2026-07-31T05:01:26.556632Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2606.09669/citation-record","integrity":"/paper/2606.09669/integrity","json":"/paper/2606.09669/citation-record.json","paper":"/paper/2606.09669"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"Introducing claude opus 4.5.https://www.anthropic.com/news/claude-opus-4-5, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:031a90cf59be05fb00c22b6959163837ffe25488a60f38854fe3e0782c8998b2","observation_id":"e19ec2f7-38ca-407d-87ce-cbeaff72d213","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"Scanqa: 3d question answering for spatial scene understanding","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:55db9be8f6a954853e5ebf8a9d90ee51d5b7b908745843868658cc6cb644e68c","observation_id":"9c66e2c6-d9d5-4130-b27c-cbd96bf48858","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"Qwen2.5-vl technical report,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:d3745020a14a824a9a53b159b08b5ec47c15990d81e6c3913d1d7c17c8c33ffa","observation_id":"b7d38452-163f-4b8e-893e-4b9bf688b039","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.13923","last_updated":"2025-02-19T18:00:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-19T18:00:14Z","title":"Qwen2.5-VL Technical Report","version":1},"cited_work":{"arxiv_id":"2502.13923","doi":"10.48550/arxiv.2502.13923","metadata_source":"pith","pith_arxiv_id":"2502.13923","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen2.5-VL Technical Report","venue":"cs.CV","work_id":"69dffacb-bfe8-442d-be86-48624c60426f","year":2025},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"cited_paper":"/paper/2502.13923","citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:98ba1929e797e1709c2979974ecf8823aaef6aeaf824d6e364815434a8e8a38d","observation_id":"f66a7bef-b29a-450e-888c-b40a0bdbe18e","resolution":{"observed_at":"2026-07-03T01:27:30.698652Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-07-12T05:19:13.082554+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T05:19:13.082554+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.06669","last_updated":"2025-08-04T04:50:21Z","snapshot_observed_at":"2026-08-04T23:08:22.516431Z","submitted_at":"2025-03-09T15:40:29Z","title":"AgiBot World Colosseo: A Large-scale Manipulation Platform for Scalable and Intelligent Embodied Systems","version":4},"cited_work":{"arxiv_id":"2503.06669","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.06669","snapshot_observed_at":"2026-07-10T23:07:47.895730Z","title":"AgiBot World Colosseo: A Large-scale Manipulation Platform for Scalable and Intelligent Embodied Systems","venue":"cs.RO","work_id":"f797e9ec-510f-43a7-8a0c-18009ce332e5","year":2025},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"cited_paper":"/paper/2503.06669","citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:11a8d8ec74096587e71a256c023840472f60f59ba228d12a91f0f31f218d6244","observation_id":"f421a06f-5405-4497-aec1-c48a90d8a332","resolution":{"observed_at":"2026-07-03T01:27:30.701092Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"Seed2.0, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:b3fa516a7b880b116d9da3bb7deeaaa07e4c694f65e0e5b30e786026bcb337a9","observation_id":"355f85ad-613a-4f62-b0ee-dd78a9934bf3","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2511.13719","doi":"10.48550/arxiv.2511.13719","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Scaling spatial intelligence with multimodal foundation models","venue":"arXiv (Cornell University)","work_id":"ceb84861-8bdb-4822-864c-8e2d0ae4d322","year":2025},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:95a8fe7f0172e9a6388def41db7c8ec4ef4edcedee5a0b732191b3741c9ea7bf","observation_id":"d520075e-3622-4749-b350-114a78798bb0","resolution":{"observed_at":"2026-07-03T01:27:30.693069Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2508.13142","doi":"10.48550/arxiv.2508.13142","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Has gpt-5 achieved spatial intelligence? an empirical study","venue":"arXiv (Cornell University)","work_id":"b0d24cae-8b82-4d58-9b26-2a803fd083cc","year":2025},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:d180ae6c991412fc0378e3f49854dfdf1112a9396f2e0ecc18ca5b97347d038c","observation_id":"f886e9f8-1cbc-460d-b9f4-95836b54f964","resolution":{"observed_at":"2026-07-03T01:27:30.696168Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"Spider2-v: How far are multi- modal agents from automating data science and engineering workflows?Advances in Neural Information Processing Systems, 37:107703–107744, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:3a539da861f59f6d5019683371ec2c1ddec93881d2602b99b5f50244b215bcef","observation_id":"b60b1f06-9489-4891-be2e-2bbb9c5e9fb1","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"Spatialvlm: Endowing vision-language models with spatial reasoning capabilities","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:9ce247af44401db1f019c9353cc19a337a2c8949c58bcaa5456ed76a474cf172","observation_id":"282f1d41-ddf1-4534-9ee6-c80143644165","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"Robogpt: an llm-based long-term decision-making embodied agent for instruction following tasks.IEEE Transactions on Cognitive and Developmental Systems, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:14743054ec541cc5ba3ecef3b8ce9d31ff1acd435d333ff59723d35c2f4c2bf7","observation_id":"70485e45-96d9-4d73-af6f-bcdc83d9b780","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.11858","last_updated":"2025-04-11T04:26:42Z","snapshot_observed_at":"2026-07-06T20:23:40.374376Z","submitted_at":"2025-01-21T03:22:10Z","title":"EmbodiedEval: Evaluate Multimodal LLMs as Embodied Agents","version":2},"cited_work":{"arxiv_id":"2501.11858","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.11858","snapshot_observed_at":"2026-07-03T11:58:05.933512Z","title":"Embodiedeval: Evaluate multimodal llms as embodied agents","venue":null,"work_id":"cb8040b1-89b3-4605-969a-34d69216c684","year":2025},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"cited_paper":"/paper/2501.11858","citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:6ef0a3363f1067c8a397292b8db707753dc45cc3da24b5f4162aacc3bb89b188","observation_id":"4cc676b1-2f72-46a1-9140-b6a3be5cac7e","resolution":{"observed_at":"2026-07-03T01:27:30.581129Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"Gemini 3 pro best for complex tasks and bringing creative concepts to life","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:dc2762d0f2e3c7704c41c3c033838382b6d03d1c77bd98f367e6a9e37678fdcf","observation_id":"95f6c51d-73b5-43a5-b907-d6a099730053","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"Proc- thor: Large-scale embodied AI using procedural generation","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:538ab566b583f6dbef1ce1a3f63388195c83a98b5845eb737d20d6fd9bec51d1","observation_id":"1c5b0d4f-ea57-4159-9c19-ba9e32c7aee4","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"Carla: An open urban driving simulator","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:844d47b3b025939beedd302105215d0bc065c339fd4e71b97026fe8176891dde","observation_id":"490fab34-15f3-4c17-bb02-7553be026abf","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"Palm-e: an embodied multimodal language model","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:f04ef496e775e373e7424ce82714bbe9342b5eac31f2296f7dd0dd0c39d936a3","observation_id":"437053b7-29fa-4d35-af72-a8c6b716f2b6","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"Embspatial-bench: Benchmarking spatial understanding for embodied tasks with large vision-language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:ff812960e574c5d1431bea09e4e4b261eaa42930bb23fe24d8faaa48afa153ba","observation_id":"ab5f3e42-3b70-4dca-82ef-2a13b1f33127","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"Vlmevalkit: An open-source toolkit for evaluating large multi-modality models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:0645f55a4aa8af4ee7259dfbabab1d9709710b796f4bd277413a27921a43c5e6","observation_id":"22d25092-d524-4f01-ab54-53ee1e75f7e7","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"Vlm-gronav: Robot naviga- tion using physically grounded vision-language models in outdoor environments","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:e0cc7220627da09b49887a08c40aa3762dd8a982b631c467e72f4cb32141c262","observation_id":"8b3900d9-8a1d-4a28-afd3-3a29b74e65a3","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"Minedojo: Building open-ended embodied agents with internet-scale knowledge.Advances in Neural Information Processing Systems, 35:18343–18362, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:38b104559360d8b7a4abcebf7531f357e410316c7453af937399e4aa4dc61745","observation_id":"bad98602-e31d-453c-a353-99be9a3793c6","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"Videoagent: A memory-augmented multimodal agent for video understanding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:8ed475a5994ed4cad3e0134cbc81da90789c5290ce06ad5fc481ff427a5c9a01","observation_id":"4a70c945-6d58-4446-8d2b-ec4e86e6c48d","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.09604","last_updated":"2024-10-12T17:49:26Z","snapshot_observed_at":"2026-08-04T16:00:02.144760Z","submitted_at":"2024-10-12T17:49:26Z","title":"EmbodiedCity: A Benchmark Platform for Embodied Agent in Real-world City Environment","version":1},"cited_work":{"arxiv_id":"2410.09604","doi":"10.48550/arxiv.2410.09604","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.09604","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2410.09604 , year=","venue":"arXiv (Cornell University)","work_id":"9fbe5487-e178-45e3-8361-22561708d2c4","year":2024},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"cited_paper":"/paper/2410.09604","citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:fc6728fc713ba1e1f775eaad2a524c7fbada3229f61e865be5ce3d89969b9433","observation_id":"99aeff15-ca31-49d5-9d8f-ec285f719cc9","resolution":{"observed_at":"2026-07-03T01:27:30.594785Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2509.06266","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T19:50:10.274403Z","title":"Spatial reasoning with vision-language models in ego-centric multi-view scenes","venue":null,"work_id":"2f4c1dea-b664-42e3-a9b6-0b1d040394a8","year":2025},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:132f65292f224a3f201f406744dfbdbb8e3963c77b2e593111cf6488ab7145ba","observation_id":"904e2b40-3fee-4344-9257-618725ee1d69","resolution":{"observed_at":"2026-07-03T01:27:30.619892Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.07062","last_updated":"2025-05-11T17:28:30Z","snapshot_observed_at":"2026-08-02T16:13:31.498470Z","submitted_at":"2025-05-11T17:28:30Z","title":"Seed1.5-VL Technical Report","version":1},"cited_work":{"arxiv_id":"2505.07062","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.07062","snapshot_observed_at":"2026-07-08T16:15:06.198778Z","title":"Seed1.5-VL Technical Report","venue":"cs.CV","work_id":"0e8e025f-ca1e-49cc-aee2-33f3a0201f3c","year":2025},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"cited_paper":"/paper/2505.07062","citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:0c1756537fa4ce8bfa1cfae1defb576ce0b406418dd639b6dcdacd5eb6cde715","observation_id":"a66b4439-5621-4308-bdb7-0f430c52ce57","resolution":{"observed_at":"2026-07-03T01:27:30.632774Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.13919","last_updated":"2024-06-06T18:37:34Z","snapshot_observed_at":"2026-07-06T17:20:11.913242Z","submitted_at":"2024-01-25T03:33:18Z","title":"WebVoyager: Building an End-to-End Web Agent with Large Multimodal Models","version":4},"cited_work":{"arxiv_id":"2401.13919","doi":"10.48550/arxiv.2401.13919","metadata_source":"pith","pith_arxiv_id":"2401.13919","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"WebVoyager: Building an End-to-End Web Agent with Large Multimodal Models","venue":"cs.CL","work_id":"12c1d840-af20-4a4e-8750-6b9c6266638f","year":2024},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"cited_paper":"/paper/2401.13919","citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:fa94e4c1bdb6bdd573671522ed1ec716237a13c0da4ea4e7d9077db2c43ba9bc","observation_id":"e558c513-fe5f-47b4-9d07-c872f0cba296","resolution":{"observed_at":"2026-07-03T01:27:30.597343Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"Cogagent: A visual language model for gui agents","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:2472c155dbaa0a8650d22ceff6b6efe153b02c348ede37c4c39739e948203b17","observation_id":"3ec8fe6e-85ed-4eea-a2c5-4ad0290399f3","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.01006","last_updated":"2026-01-01T13:07:25Z","snapshot_observed_at":"2026-08-03T18:50:30.558321Z","submitted_at":"2025-07-01T17:55:04Z","title":"GLM-4.5V and GLM-4.1V-Thinking: Towards Versatile Multimodal Reasoning with Scalable Reinforcement Learning","version":6},"cited_work":{"arxiv_id":"2507.01006","doi":"10.48550/arxiv.2507.01006","metadata_source":"pith","pith_arxiv_id":"2507.01006","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GLM-4.5V and GLM-4.1V-Thinking: Towards Versatile Multimodal Reasoning with Scalable Reinforcement Learning","venue":"cs.CV","work_id":"366607ba-e4ea-4726-98c3-63356e32351c","year":2025},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"cited_paper":"/paper/2507.01006","citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:432671d684789aa31bc08a2cc3bc1501ec02418ed83a23237c2eff507e897c14","observation_id":"f58e2aa4-ffbb-43da-b36e-23e0596ffde8","resolution":{"observed_at":"2026-07-03T01:27:30.638601Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"3d concept learning and reasoning from multi-view images","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:9766f0de028f5324b528ea3dd314a2f4ca41a12115f0b33663db8eb23ec9f928","observation_id":"201398b8-5fd3-43ff-9d82-33be8351ed0d","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"Gqa: A new dataset for real-world visual reasoning and compositional question answering","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:a9a550194f838799d9daf4b67eed67a9280d72d60e890d42736a755ebd89657d","observation_id":"23405645-c976-4165-bf18-35fe376620f6","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"OmniSpatial: Towards comprehensive spatial reasoning benchmark for vision language models","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:c45f7e6e763089f44c5ac7bbca65ac4d5f5713e7598ccfbfaed7269269b28db5","observation_id":"c0a17fe4-11fb-4d73-9ca2-553efb806878","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"Clevr: A diagnostic dataset for compositional language and elementary visual reasoning","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:ab92110de6b826c65eb240493f3f5d224b18268a4d2213f6ae33ec9b9cdc5615","observation_id":"0ba344a5-5b0b-45f8-adc4-bea208ebcc72","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"Visualwebarena: Evaluating multimodal agents on realistic visual web tasks","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:aa5080dc0af42cc81eb2f2a71cbd2107a8b6e0a9d9f5d22b089fcc1cccbbffef","observation_id":"0e36c439-f820-4daf-88cc-a87165ded012","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1712.05474","last_updated":"2022-08-26T17:12:17Z","snapshot_observed_at":"2026-07-06T06:14:28.435222Z","submitted_at":"2017-12-14T23:17:24Z","title":"AI2-THOR: An Interactive 3D Environment for Visual AI","version":4},"cited_work":{"arxiv_id":"1712.05474","doi":"10.48550/arxiv.1712.05474","metadata_source":"pith","pith_arxiv_id":"1712.05474","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"AI2-THOR: An Interactive 3D Environment for Visual AI","venue":"cs.CV","work_id":"9c86ed28-ea70-424c-bd56-34f59dcad861","year":2017},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"cited_paper":"/paper/1712.05474","citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:3503cd0f30086dbeb19a71be55a5c40c2eef2b75e935fdabd41591f34283f095","observation_id":"a0271b8b-b6f8-42c8-89e6-5a7b5f579fab","resolution":{"observed_at":"2026-07-03T01:27:30.665294Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.14030","last_updated":"2025-05-29T01:50:49Z","snapshot_observed_at":"2026-08-03T05:45:41.256169Z","submitted_at":"2025-05-20T07:29:26Z","title":"AutoBio: A Simulation and Benchmark for Robotic Automation in Digital Biology Laboratory","version":3},"cited_work":{"arxiv_id":"2505.14030","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.14030","snapshot_observed_at":"2026-07-10T02:06:42.137652Z","title":"Autobio: A simulation and benchmark for robotic automation in digital biology laboratory","venue":"cs.RO","work_id":"5cffd093-d41c-4ed4-bafe-1a9999e6ab0f","year":2025},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"cited_paper":"/paper/2505.14030","citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:f24e7b7986b6eff2e929bff93560f92c87f6fc45f9546e96f026f4cf2fa42138","observation_id":"c5769039-0a8d-443a-be16-af765d62c9b4","resolution":{"observed_at":"2026-07-03T01:27:30.574810Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2108.03272","last_updated":"2021-11-03T18:51:07Z","snapshot_observed_at":"2026-07-06T11:36:17.714076Z","submitted_at":"2021-08-06T18:41:39Z","title":"iGibson 2.0: Object-Centric Simulation for Robot Learning of Everyday Household Tasks","version":4},"cited_work":{"arxiv_id":"2108.03272","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2108.03272","snapshot_observed_at":"2026-07-04T17:49:59.925861Z","title":"igibson 2.0: Object-centric simulation for robot learning of everyday household tasks","venue":null,"work_id":"e7427c0a-48e2-4cff-81e8-e2063f12e1da","year":2021},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"cited_paper":"/paper/2108.03272","citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:94975fd3784142da77bf023c57c2275230e29a6b82c328a6eb864907947a81c2","observation_id":"d883ac61-6992-4006-8511-d1557f1e8af1","resolution":{"observed_at":"2026-07-03T01:27:30.653610Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07166","last_updated":"2025-01-19T19:29:50Z","snapshot_observed_at":"2026-07-06T19:30:34.984716Z","submitted_at":"2024-10-09T17:59:00Z","title":"Embodied Agent Interface: Benchmarking LLMs for Embodied Decision Making","version":3},"cited_work":{"arxiv_id":"2410.07166","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2410.07166","snapshot_observed_at":"2026-07-03T01:27:30.603500Z","title":"Bill Yuchen Lin, Yicheng Fu, Karina Yang, Faeze Brah- man, Shiyu Huang, Chandra Bhagavatula, Prithviraj Ammanabrolu, Yejin Choi, and Xiang Ren","venue":null,"work_id":"2adbc76b-a0ec-4149-b8e5-852995c7cc88","year":2024},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"cited_paper":"/paper/2410.07166","citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:e857ffba1da7e0d9176c44b2586550f734130b0ad6e7df15aa2b335a9efc7c85","observation_id":"5b0fd86d-b567-49ba-a5e0-9faa1c78e798","resolution":{"observed_at":"2026-07-03T01:27:30.604976Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.10763","last_updated":"2023-12-17T16:53:30Z","snapshot_observed_at":"2026-08-02T02:02:01.525485Z","submitted_at":"2023-12-17T16:53:30Z","title":"M3DBench: Let's Instruct Large Models with Multi-modal 3D Prompts","version":1},"cited_work":{"arxiv_id":"2312.10763","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2312.10763","snapshot_observed_at":"2026-07-04T04:29:35.010933Z","title":"Rong Li, Shijie Li, Lingdong Kong, Xulei Yang, and Junwei Liang","venue":null,"work_id":"d435f4e6-882b-4bc6-a5c1-66b32f32fa7d","year":2025},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"cited_paper":"/paper/2312.10763","citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:a11977172b423e0b0577faeec29d7db84dc264d0e7b07f1d5d8442ed25fdfae6","observation_id":"33d6404e-0cbb-4f70-9f7c-e847df3e248b","resolution":{"observed_at":"2026-07-03T01:27:30.616917Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"Metadrive: Composing diverse driving scenarios for generalizable reinforcement learning","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:75bfa750071e50af05719a803dcb57b7af51b20651a63c06a99d395b82c654fa","observation_id":"4493eb09-8c01-489f-86f2-0d0d2b90c1a2","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.17419","last_updated":"2025-06-25T02:24:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-24T18:50:52Z","title":"From System 1 to System 2: A Survey of Reasoning Large Language Models","version":6},"cited_work":{"arxiv_id":"2502.17419","doi":"10.48550/arxiv.2502.17419","metadata_source":"pith","pith_arxiv_id":"2502.17419","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"From System 1 to System 2: A Survey of Reasoning Large Language Models","venue":"cs.AI","work_id":"67495af7-38a4-40b1-bd10-fd9d939c7abd","year":2025},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"cited_paper":"/paper/2502.17419","citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:5750d7f3cfef9c0e8656f0b948c6e69398ad1c23914d858f60ba55d8e3a488d9","observation_id":"84dc1d62-ee12-427a-a3da-6415615ffdf3","resolution":{"observed_at":"2026-07-03T01:27:30.592032Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"Vila: On pre-training for visual language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:37f4dddf8870f7a45cba8d8a32b47df3d7e0f6ae5f376e1298699e94a244c8fc","observation_id":"fadd6697-439e-4a60-9b06-9cd7213e8244","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2512.10863","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T10:48:03.213683Z","title":"Mmsi-video-bench: A holistic benchmark for video-based spatial intelligence","venue":null,"work_id":"5576e252-6316-45fb-a511-bcbf65609546","year":2025},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:0322d6bad0c91044cc92a1bdc669641b5d434be4827c4312eef073e73c256996","observation_id":"9cca9819-9e55-453f-830a-1775486ac4df","resolution":{"observed_at":"2026-07-03T01:27:30.676637Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2507.07984","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T01:27:30.587936Z","title":"Ost-bench: Evaluating the capabilities of mllms in online spatio-temporal scene understanding","venue":null,"work_id":"ca782bad-f2ab-48af-9cca-a30e5aebf8e2","year":2025},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:0bb882cb3a0f47cf2ea553706b354d64e9ff553feebd4f0f82ecc2cbe180a38f","observation_id":"7faaa7df-a02b-4549-9973-be32c17c1caa","resolution":{"observed_at":"2026-07-03T01:27:30.589508Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"Visual spatial reasoning.Transactions of the Association for Computational Linguistics, 11:635–651, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:948f4aa323bcad3ba4f63d8ce656823dd733d8d29b4f039e1e50055991cad566","observation_id":"390a2ffa-94be-478f-b236-e0d49b6011c1","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"Llava-plus: Learning to use tools for creating multimodal agents","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:1ebdc3c0bcd28a62f288a5a8e1a145bfe9e76d36e73979d39ac327d0d10b2998","observation_id":"9409299b-06ca-4713-86bb-867596d94f2e","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2511.15722","doi":"10.48550/arxiv.2511.15722","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Spatial reasoning in multimodal large language models: A survey of tasks, benchmarks and methods","venue":"ArXiv.org","work_id":"6a89950e-e68f-4ee1-92ca-4c617487ec7c","year":2025},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:1e3f063e3b79f2c5b90d2228286a990067f4b7a8fdd8edf7f2558a14f1281b32","observation_id":"4463b37a-7455-440d-b0ce-ba3746587120","resolution":{"observed_at":"2026-07-03T01:27:30.687723Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.10074","last_updated":"2025-01-23T02:31:25Z","snapshot_observed_at":"2026-07-06T20:22:23.609336Z","submitted_at":"2025-01-17T09:46:27Z","title":"SpatialCoT: Advancing Spatial Reasoning through Coordinate Alignment and Chain-of-Thought for Embodied Task Planning","version":3},"cited_work":{"arxiv_id":"2501.10074","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2501.10074","snapshot_observed_at":"2026-07-03T22:28:59.692534Z","title":"Spatialcot: Advancing spatial reasoning through coordinate alignment and chain-of-thought for embodied task planning.arXiv preprint arXiv:2501.10074, 2025a","venue":null,"work_id":"0862e22b-0037-4b5d-9f96-5ec002097396","year":2025},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"cited_paper":"/paper/2501.10074","citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:3383e52ae37abe06c4ffb68399fdfa585f97edf4b69204699f5a850b8a7d5416","observation_id":"8cc528be-b179-4200-b688-23ea88a620e4","resolution":{"observed_at":"2026-07-03T01:27:30.607933Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"3DSRBench: A comprehensive 3D spatial reasoning benchmark","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:aa9581d0ac37571b5aeb7d5ff5dc3ff7198feee19dbd8aae0ebd727a6d738903","observation_id":"2edc5e87-a03e-4724-8dd3-17f7e341d694","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.07474","last_updated":"2023-04-12T20:05:41Z","snapshot_observed_at":"2026-07-06T14:05:02.813765Z","submitted_at":"2022-10-14T02:52:26Z","title":"SQA3D: Situated Question Answering in 3D Scenes","version":5},"cited_work":{"arxiv_id":"2210.07474","doi":null,"metadata_source":"pith","pith_arxiv_id":"2210.07474","snapshot_observed_at":"2026-07-11T01:37:43.684631Z","title":"Sqa3d: Situated question answering in 3d scenes","venue":"cs.CV","work_id":"8d9797f2-5882-4b0e-99d7-62e075c67387","year":2022},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"cited_paper":"/paper/2210.07474","citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:710ced11322fc1bd56829c39ced2ba4edc38422b3a475840265ffb9ab9c1d57a","observation_id":"e3bbb6fe-fb9b-44d0-8d3b-509856a63db1","resolution":{"observed_at":"2026-07-03T01:27:30.647837Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"Introducing gpt-5.2.https://openai.com/index/introducing-gpt-5-2/, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:8fa5d29580b81a6003eaea5a4b5449ef9d4c3d9edb82510d92fdaa806b71c973","observation_id":"52e7fa26-025b-4aa9-b779-206c23cbe5ae","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"Gpt -5.4 thinking system card, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:c75b61b246ca17fa55c6bf9b264d5dcaf3cfe6adf5550d73e9bf3d46d3c1bb54","observation_id":"7fc78741-d1e9-457b-8856-4f4cdde1eb20","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2018.00886","doi":"10.1109/cvpr.2018.00886","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"VirtualHome: SimulatingHouseholdActivitiesViaPrograms","venue":null,"work_id":"edb32d95-0a38-44d9-bf5e-7cacd8568531","year":2018},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:f15630db865e04577fee04877db85335993185edcb20c2293ccccca1fc9e96ad","observation_id":"e57cfd88-4ad6-49ce-b2c1-0c25249a8f5d","resolution":{"observed_at":"2026-06-27T16:41:03.041882Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2109.08238","last_updated":"2021-09-16T22:01:24Z","snapshot_observed_at":"2026-07-06T11:48:35.206160Z","submitted_at":"2021-09-16T22:01:24Z","title":"Habitat-Matterport 3D Dataset (HM3D): 1000 Large-scale 3D Environments for Embodied AI","version":1},"cited_work":{"arxiv_id":"2109.08238","doi":"10.48550/arxiv.2501.01366","metadata_source":"pith","pith_arxiv_id":"2109.08238","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Habitat-Matterport 3D Dataset (HM3D): 1000 Large-scale 3D Environments for Embodied AI","venue":"cs.CV","work_id":"fd1d6aaa-d036-4baf-9601-1435c8cefb37","year":2021},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"cited_paper":"/paper/2109.08238","citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:c18cc5366df2c4a21adb9d0a26261a82ac15a8cb3185179022c2b4df4d06b5c2","observation_id":"716a930f-e5ec-463b-87f8-6fa3ecfa264c","resolution":{"observed_at":"2026-07-03T01:27:30.670960Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.14573","last_updated":"2025-04-06T20:37:50Z","snapshot_observed_at":"2026-07-06T18:18:39.093732Z","submitted_at":"2024-05-23T13:48:54Z","title":"AndroidWorld: A Dynamic Benchmarking Environment for Autonomous Agents","version":5},"cited_work":{"arxiv_id":"2405.14573","doi":"10.48550/arxiv.2405.14573","metadata_source":"pith","pith_arxiv_id":"2405.14573","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"AndroidWorld: A Dynamic Benchmarking Environment for Autonomous Agents","venue":"cs.AI","work_id":"c5116d19-d3d3-40fd-9620-f7489812a9ba","year":2024},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"cited_paper":"/paper/2405.14573","citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:9db58d799c77b6c065974bb0b360b41dab88284526bf0bf6702eb9d614924fc6","observation_id":"8be52b7c-ecef-4d84-97a5-41d585b2d87f","resolution":{"observed_at":"2026-07-03T01:27:30.659784Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"Habitat: A platform for embodied ai research","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:04caeb773e73461d93025a3cf74d092c84cef1a22c4d2acf91c2645d0f59cadf","observation_id":"f5e1b9ed-7880-456f-8f2f-c6e270ddbd85","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"ALFRED: A benchmark for interpreting grounded instructions for everyday tasks","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:446a834e949f176947e1ab33f75d2a31233ddbc1350979d29702cd10fc3a97ac","observation_id":"7d6136a1-e4e2-4bf0-b42c-b023e1276659","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2601.03267","last_updated":"2026-05-01T23:55:43Z","snapshot_observed_at":"2026-08-02T10:52:10.211700Z","submitted_at":"2025-12-19T07:05:38Z","title":"OpenAI GPT-5 System Card","version":2},"cited_work":{"arxiv_id":"2601.03267","doi":"10.48550/arxiv.2601.03267","metadata_source":"pith","pith_arxiv_id":"2601.03267","snapshot_observed_at":"2026-08-05T02:49:54.815029Z","title":"OpenAI GPT-5 System Card","venue":"cs.CL","work_id":"ca87689a-0d29-4476-b504-b65dbbb08af4","year":2025},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"cited_paper":"/paper/2601.03267","citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:423ee6e06d8dbfb846ca068a06e646c6ed7b349212606aef325c5533d4f3aa28","observation_id":"730ebfde-bd4a-4a2d-912c-fb186366aa0d","resolution":{"observed_at":"2026-07-03T01:27:30.684772Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-03T00:38:11.458508+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T00:38:11.458508+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2512.18028","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T01:27:30.633898Z","title":"Corso, and Eric Sax","venue":null,"work_id":"4085ae70-13b9-403d-afe9-18a6c6a04835","year":2025},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:0e5319d8c073ad4eb8912f19f33cb575e3990f4ff59a2b4747d57099c0f7fb32","observation_id":"37545a29-3088-4d74-88eb-b82e4ada0810","resolution":{"observed_at":"2026-07-03T01:27:30.635879Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"Gemini 3 pro: the frontier of vision ai, 2025b","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:2909262b3e1da2642cea023d8c30afaef4826ebbe11d382dbf05cb4d22e0728c","observation_id":"261c5356-9983-4d93-b8bb-a3696dff785a","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"Gemini 3 flash, 2025b","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:8f2c8b66630b0652811105faaeac19f80c06a3b509d7ea71e147b5f26af9f19b","observation_id":"7e6567fb-826a-42d5-bf77-390d819a8e1a","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":"2312.11805","doi":"10.1038/nrn2888","metadata_source":"pith","pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gemini: A Family of Highly Capable Multimodal Models","venue":"cs.CL","work_id":"83f7c85b-3f11-450f-ac0c-64d9745220b2","year":2023},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:b8b3bbe4f8a80cf4049d79f98cbba80b9f3ff4e46df48d7307372712b21981da","observation_id":"682748b4-3303-4ab4-969c-23004fdc38be","resolution":{"observed_at":"2026-07-03T01:27:30.650444Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.06261","last_updated":"2025-12-19T14:25:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-07T17:36:04Z","title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","version":6},"cited_work":{"arxiv_id":"2507.06261","doi":"10.48550/arxiv.2503.19","metadata_source":"pith","pith_arxiv_id":"2507.06261","snapshot_observed_at":"2026-07-11T03:17:51.364436Z","title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","venue":"cs.CL","work_id":"008df105-2fdd-45d8-857a-8e35868aecb6","year":2025},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"cited_paper":"/paper/2507.06261","citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:cc657f4d5b5ded075db299c80ead1769ed2788e493681ff32f7a1c6931b30dc9","observation_id":"0f504f61-0745-49ee-a3b9-7063321a9ba5","resolution":{"observed_at":"2026-07-03T01:27:30.690088Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"Glm-4.6v: Open source multimodal models with native tool use, 2025a","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:cbc4f21eb2b84177d65e7e206c2c6b7331974f16da744c8c4c73bf034315e18d","observation_id":"3e1e8a3a-1162-4726-89c9-471d29186c50","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.07491","last_updated":"2025-06-23T13:45:50Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-10T06:48:26Z","title":"Kimi-VL Technical Report","version":3},"cited_work":{"arxiv_id":"2504.07491","doi":null,"metadata_source":"pith","pith_arxiv_id":"2504.07491","snapshot_observed_at":"2026-07-08T06:34:41.884360Z","title":"Kimi-VL Technical Report","venue":"cs.CV","work_id":"c876520f-8a20-44f3-b92a-bf7d35bd430f","year":2025},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"cited_paper":"/paper/2504.07491","citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:4cdb41dc7929244b17c2bbd25d795f2b8480b10e914ac0338564a734e9f4cfc2","observation_id":"c6e77ae3-82e5-426a-bb9f-e5f6c8b2ce78","resolution":{"observed_at":"2026-07-03T01:27:30.679664Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2602.02276","last_updated":"2026-02-02T16:17:38Z","snapshot_observed_at":"2026-07-06T22:44:09.804048Z","submitted_at":"2026-02-02T16:17:38Z","title":"Kimi K2.5: Visual Agentic Intelligence","version":1},"cited_work":{"arxiv_id":"2602.02276","doi":"10.48550/arxiv.2602.02276","metadata_source":"pith","pith_arxiv_id":"2602.02276","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Kimi K2.5: Visual Agentic Intelligence","venue":"cs.CL","work_id":"d690be8f-5d53-49b0-b1e7-79668eb8fcdb","year":2026},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"cited_paper":"/paper/2602.02276","citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:537265492d3eac049a7e2c4678553897d8724d1c4111cb74e99a1a95247988de","observation_id":"608bc7a3-0356-43cf-9214-46a9e2e05c9d","resolution":{"observed_at":"2026-07-03T01:27:30.673556Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"Qwen3.5: Accelerating productivity with native multimodal agents, February","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:065ea422afe530a2fdaa68d6bff6dbe717a6df3ece8db4a3535909236cca4aa2","observation_id":"76141400-ff9a-43f4-b460-a978f86dd3ba","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:1e337881300485d05b8b2fadb4187ac061bd0c3ae90e92f4ed94a4bb6a907394","observation_id":"03f265a7-e397-4ef6-9d3d-6aed2eec7870","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2509.02544","last_updated":"2025-09-05T14:59:27Z","snapshot_observed_at":"2026-08-05T09:41:26.544360Z","submitted_at":"2025-09-02T17:44:45Z","title":"UI-TARS-2 Technical Report: Advancing GUI Agent with Multi-Turn Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2509.02544","doi":"10.1037/xlm0000535","metadata_source":"pith","pith_arxiv_id":"2509.02544","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"UI-TARS-2 Technical Report: Advancing GUI Agent with Multi-Turn Reinforcement Learning","venue":"cs.AI","work_id":"422846c6-e6e2-47c9-9065-85cc09c07cd6","year":2025},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"cited_paper":"/paper/2509.02544","citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:1ac7d02530506ccbe90c8cf1dadfd079e63b2b1a02bea6711436c0dedded7f60","observation_id":"3b536508-5ced-4349-901c-1d1cfa4170e4","resolution":{"observed_at":"2026-07-03T01:27:30.662721Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"Is a picture worth a thousand words? delving into spatial reasoning for vision language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:7417fd9a7fb71ca7b88d7e64735c9d49883b1c04c86add94365a437b6ef2ea8c","observation_id":"fc543fad-a748-4ef5-882d-5a3642b2d122","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.16158","last_updated":"2024-04-18T06:53:38Z","snapshot_observed_at":"2026-08-05T03:22:24.562938Z","submitted_at":"2024-01-29T13:46:37Z","title":"Mobile-Agent: Autonomous Multi-Modal Mobile Device Agent with Visual Perception","version":2},"cited_work":{"arxiv_id":"2401.16158","doi":"10.48550/arxiv.2401.16158","metadata_source":"pith","pith_arxiv_id":"2401.16158","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mobile-Agent: Autonomous Multi-Modal Mobile Device Agent with Visual Perception","venue":"cs.CL","work_id":"c8f4a2c0-2743-4264-acca-86fbf64ed3ba","year":2024},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"cited_paper":"/paper/2401.16158","citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:3fc3716b0478f9c6963f76f4389d400b3435460377ebabb3ee07c786d556073e","observation_id":"435b9a5a-a0f0-486c-b535-5ecca914db45","resolution":{"observed_at":"2026-07-03T01:27:30.628006Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2312.09245","doi":"10.48550/arxiv.2312.09245","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Drivemlm: Aligning multi-modal large language models with behavioral planning states for au- tonomous driving","venue":"arXiv (Cornell University)","work_id":"9410f6b5-aea2-4ee3-a94a-b31be4433519","year":2023},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:f4af4bcf04283447fe838902c965dba6d49aa7cc765569af9b4d416039c3caa0","observation_id":"21213f33-8e68-4f38-aa59-41b378b34bf4","resolution":{"observed_at":"2026-07-03T01:27:30.668210Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-07-09T19:19:07.387977+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-09T19:19:07.387977+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"SITE: Towards spatial intelligence thorough evaluation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:de4755a55cd1584ccd23b7dee41ecd1eaadeb3db73ec11a857f1a53162063088","observation_id":"4966711e-80b7-4215-ae12-4ddab166b700","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2508.09123","doi":"10.48550/arxiv.2508.09123","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mobile-agent-v2: Mobile device operation assistant with effective navigation via multi-agent collaboration.Advances in Neural Information Processing Systems, 37:2686–2710, 2024a","venue":"arXiv (Cornell University)","work_id":"1648be89-5611-4615-8ef6-f051a980641c","year":2025},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:3b05f04a88c7138df8069f7d233543c173d562fd1859c9c6ef82a068107c616f","observation_id":"b95a30a8-a914-455c-a66f-bed685e87999","resolution":{"observed_at":"2026-07-03T01:27:30.602431Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.17012","last_updated":"2026-04-13T12:33:41Z","snapshot_observed_at":"2026-07-06T21:28:43.610849Z","submitted_at":"2025-05-22T17:59:03Z","title":"SpatialScore: Towards Comprehensive Evaluation for Spatial Intelligence","version":3},"cited_work":{"arxiv_id":"2505.17012","doi":"10.48550/arxiv.2505.17012","metadata_source":"pith","pith_arxiv_id":"2505.17012","snapshot_observed_at":"2026-08-05T02:49:54.815029Z","title":"SpatialScore: Towards Comprehensive Evaluation for Spatial Intelligence","venue":"cs.CV","work_id":"7fe32ac0-2fff-4a5e-958f-28092c0ee055","year":2025},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"cited_paper":"/paper/2505.17012","citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:9512cec28de1e7af864d328722728c2b42a9d7112dce5fd4a723379342ba677e","observation_id":"3c35a0f5-bc28-4e26-a091-49f7a7036520","resolution":{"observed_at":"2026-07-03T01:27:30.583555Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"Gibson env: Real-world perception for embodied agents","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:6ae2804158190bb18bb0e09dc7943711745c871d51a950f373c4087f4a66a0ce","observation_id":"30b9fc91-27f8-4b12-8a65-e9f40ce592be","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"Sapien: A simulated part-based interactive environ- ment","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:e44edb70ed0bf96980e2b6e0a10a97ecb6dcaa42cf3b5f40c99714ba346ab123","observation_id":"9384b297-522b-47a9-b872-689bf72615bb","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.15116","last_updated":"2024-02-23T06:04:23Z","snapshot_observed_at":"2026-07-06T17:34:24.337596Z","submitted_at":"2024-02-23T06:04:23Z","title":"Large Multimodal Agents: A Survey","version":1},"cited_work":{"arxiv_id":"2402.15116","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2402.15116","snapshot_observed_at":"2026-07-03T09:07:48.049036Z","title":"Large multimodal agents: A survey","venue":null,"work_id":"5f1eddc5-e861-484d-a204-926b2bfd8ecc","year":2024},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"cited_paper":"/paper/2402.15116","citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:66acb2c6e48b4fb53ad360b87e91dd962fb2a4f8838fbede964b607f3825e936","observation_id":"a0fa33ef-b1e2-44be-a90e-0a4a59c37316","resolution":{"observed_at":"2026-07-03T01:27:30.599891Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"Osworld: Benchmarking multimodal agents for open-ended tasks in real computer environments.Advances in Neural Information Processing Systems, 37:52040–52094, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:25e7bedc0eb48d5148b251d2214e93d5867a19f8b6cf103890f65437e8c74243","observation_id":"23390021-1232-4d1f-8378-e34d27642b40","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2511.21471","last_updated":"2026-05-07T07:59:46Z","snapshot_observed_at":"2026-08-02T23:27:20.204280Z","submitted_at":"2025-11-26T15:04:18Z","title":"SpatialBench: Benchmarking Multimodal Large Language Models for Spatial Cognition","version":4},"cited_work":{"arxiv_id":"2511.21471","doi":null,"metadata_source":"pith","pith_arxiv_id":"2511.21471","snapshot_observed_at":"2026-07-04T08:59:43.334805Z","title":"SpatialBench: Benchmarking Multimodal Large Language Models for Spatial Cognition","venue":"cs.AI","work_id":"81942f57-c90d-41fc-9036-779f35d52bed","year":2025},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"cited_paper":"/paper/2511.21471","citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:37c6494acf7598cbb77008c1da10a8988b297e2a1993b1ad5fa1d9946dce2f1d","observation_id":"8f60c5b8-8177-45be-a72d-bd80e57f6758","resolution":{"observed_at":"2026-07-03T01:27:30.644667Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"Pointllm: Empowering large language models to understand point clouds","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:1a94ce2cabf470377db285d188f401a783a791d21a0abe90db00a3c5299c9109","observation_id":"f8eeb98d-69b3-4123-b257-cc669f1713e9","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"cited_work":{"arxiv_id":"2505.09388","doi":"10.1016/j.aiopen.2022.12","metadata_source":"pith","pith_arxiv_id":"2505.09388","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Qwen3 Technical Report","venue":"cs.CL","work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e","year":2025},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"cited_paper":"/paper/2505.09388","citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:dbd6211de77b3507900d3ff8ad06d227aafe19d3052d59096f3899684374c781","observation_id":"5e37bb14-aaaf-485c-837b-d1260cde98ac","resolution":{"observed_at":"2026-07-03T01:27:30.656994Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"Thinking in space: How multimodal large language models see, remember, and recall spaces","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:a0eb9993a2ec8e5145bd639b1e09121d3aba083103cabf2a19de2a6468eeebf9","observation_id":"16fb1da4-b107-451d-a9e8-9f658e839d15","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.09560","last_updated":"2025-06-05T07:22:50Z","snapshot_observed_at":"2026-07-06T20:36:12.548343Z","submitted_at":"2025-02-13T18:11:34Z","title":"EmbodiedBench: Comprehensive Benchmarking Multi-modal Large Language Models for Vision-Driven Embodied Agents","version":3},"cited_work":{"arxiv_id":"2502.09560","doi":"10.48550/arxiv.2502.09560","metadata_source":"pith","pith_arxiv_id":"2502.09560","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"EmbodiedBench: Comprehensive Benchmarking Multi-modal Large Language Models for Vision-Driven Embodied Agents","venue":"cs.AI","work_id":"b1e694c6-fe5b-477f-ab8f-801e0fb0412f","year":2025},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"cited_paper":"/paper/2502.09560","citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:2879c884a868bdf5358a5ca1a1760b76c429002ef222b85c8617accb178926b1","observation_id":"e4bdfe65-5a21-4455-9f69-007194d073a2","resolution":{"observed_at":"2026-07-03T01:27:30.613186Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.23764","last_updated":"2026-05-23T03:42:56Z","snapshot_observed_at":"2026-07-06T21:33:07.415628Z","submitted_at":"2025-05-29T17:59:52Z","title":"MMSI-Bench: A Benchmark for Multi-Image Spatial Intelligence","version":3},"cited_work":{"arxiv_id":"2505.23764","doi":"10.48550/arxiv.2505.23764","metadata_source":"pith","pith_arxiv_id":"2505.23764","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mmsi-bench: A benchmark for multi-image spatial intelligence","venue":"cs.CV","work_id":"3c86b01a-2b7e-4a70-94c5-92bf4d05ad52","year":2025},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"cited_paper":"/paper/2505.23764","citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:7ef9b44ba6eb97010fe16d4bc9911dbe5b8c5cb048928b106fb7a61d7fdf5316","observation_id":"9c0d1755-532a-4557-9ae5-bda7c82698d9","resolution":{"observed_at":"2026-07-03T01:27:30.641550Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"Drivearena: A closed-loop generative simulation platform for autonomous driving","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:0dea37d3a47f99d9e761d2f0f36d0d4ff1175b36283cc4abf8b5785bb6fdb614","observation_id":"1cca129d-e0bc-4fd0-8e6c-0255cea06803","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"Meta-world: A benchmark and evaluation for multi-task and meta reinforcement learning","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:425913de66ad32f0b2306bc895b1224bd8ca577e6dfff72effeff0243ea2c4cb","observation_id":"cffc97cb-c8d0-412c-af02-d858758f39c9","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.13601","last_updated":"2024-05-28T05:36:23Z","snapshot_observed_at":"2026-07-06T17:20:00.101661Z","submitted_at":"2024-01-24T17:10:45Z","title":"MM-LLMs: Recent Advances in MultiModal Large Language Models","version":5},"cited_work":{"arxiv_id":"2401.13601","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.13601","snapshot_observed_at":"2026-07-04T15:09:55.103963Z","title":"Mm-llms: Recent ad- vances in multimodal large language models","venue":null,"work_id":"4c35990c-f0e6-4be5-bdd6-e77e3eaddf23","year":2024},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"cited_paper":"/paper/2401.13601","citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:78f78983398f608d6d210047e200b3be7196096f9e5307adac90081823a32cd7","observation_id":"3ddff7b5-55cb-44d7-8fe0-8bc68c1087ed","resolution":{"observed_at":"2026-07-03T01:27:30.610598Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.18194","last_updated":"2024-12-24T06:03:42Z","snapshot_observed_at":"2026-08-02T00:03:22.878669Z","submitted_at":"2024-12-24T06:03:42Z","title":"VLABench: A Large-Scale Benchmark for Language-Conditioned Robotics Manipulation with Long-Horizon Reasoning Tasks","version":1},"cited_work":{"arxiv_id":"2412.18194","doi":"10.48550/arxiv.2412.18194","metadata_source":"pith","pith_arxiv_id":"2412.18194","snapshot_observed_at":"2026-07-10T06:15:00.866473Z","title":"Vlabench: A large-scale benchmark for language-conditioned robotics manipulation with long-horizon reasoning tasks","venue":"cs.RO","work_id":"9ce9e004-2bc2-433e-82c4-460eddd18830","year":2024},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"cited_paper":"/paper/2412.18194","citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:7b61e959c155040340659a96845010c5364f12ee5723036ba1ed7046e367a576","observation_id":"26ed2b1d-1f73-49cf-9e10-c90681737d50","resolution":{"observed_at":"2026-07-03T01:27:30.622739Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.11630","last_updated":"2025-08-15T17:59:49Z","snapshot_observed_at":"2026-08-03T04:02:35.392835Z","submitted_at":"2025-08-15T17:59:49Z","title":"Thyme: Think Beyond Images","version":1},"cited_work":{"arxiv_id":"2508.11630","doi":"10.48550/arxiv.2508.11630","metadata_source":"pith","pith_arxiv_id":"2508.11630","snapshot_observed_at":"2026-07-11T03:17:52.050556Z","title":"Thyme: Think Beyond Images","venue":"cs.CV","work_id":"f91f31cb-6ce5-43a8-b71e-9fc90a2b4160","year":2025},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"cited_paper":"/paper/2508.11630","citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:cbccbcdb7b04a3591c2a21126295b888312b9999184c487bd10ffcae5d5dd2e9","observation_id":"2dd119b8-baaf-4135-a0e9-9f9daf40e8aa","resolution":{"observed_at":"2026-07-03T01:27:30.578474Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.14362","last_updated":"2026-03-01T04:59:56Z","snapshot_observed_at":"2026-08-02T12:23:34.946873Z","submitted_at":"2025-05-20T13:48:11Z","title":"DeepEyes: Incentivizing \"Thinking with Images\" via Reinforcement Learning","version":3},"cited_work":{"arxiv_id":"2505.14362","doi":"10.48550/arxiv.2505.14362","metadata_source":"pith","pith_arxiv_id":"2505.14362","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DeepEyes: Incentivizing \"Thinking with Images\" via Reinforcement Learning","venue":"cs.CV","work_id":"5f6cf57b-2407-4127-b39c-d8a61494e474","year":2025},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"cited_paper":"/paper/2505.14362","citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:067b0752ea008dac75d105e678edddea65e0af526d98b9701f195aa44b680a99","observation_id":"675f39fe-051c-432c-ad1a-7f5a508eb2c5","resolution":{"observed_at":"2026-07-03T01:27:30.625350Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2512.13660","last_updated":"2026-07-03T09:12:21Z","snapshot_observed_at":"2026-08-03T16:27:22.621857Z","submitted_at":"2025-12-15T18:52:43Z","title":"Towards Spatial Trace with Reasoning in Vision-Language Models for Robotics","version":4},"cited_work":{"arxiv_id":"2512.13660","doi":null,"metadata_source":"pith","pith_arxiv_id":"2512.13660","snapshot_observed_at":"2026-07-04T19:50:10.297225Z","title":"Towards Spatial Trace with Reasoning in Vision-Language Models for Robotics","venue":"cs.RO","work_id":"b45ee402-2831-4993-b4c6-37dbef7b4070","year":2025},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"cited_paper":"/paper/2512.13660","citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:0767dd514cbfeb17b0af1180397904a4830bfab6a8354f06a68ae964e1c76501","observation_id":"976c17db-01b1-48af-825c-55d53e007d73","resolution":{"observed_at":"2026-07-03T01:27:30.630444Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.18125","last_updated":"2025-04-27T06:50:23Z","snapshot_observed_at":"2026-07-06T19:22:58.341345Z","submitted_at":"2024-09-26T17:59:11Z","title":"LLaVA-3D: A Simple yet Effective Pathway to Empowering LMMs with 3D-awareness","version":3},"cited_work":{"arxiv_id":"2409.18125","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2409.18125","snapshot_observed_at":"2026-07-04T15:39:56.857028Z","title":"Llava-3d: A simple yet effective pathway to empowering lmms with 3d-awareness","venue":null,"work_id":"25557a62-345c-4dfe-a582-4ff77ad59e6a","year":2024},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"cited_paper":"/paper/2409.18125","citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:e7cfaafc392745d95b9cda236d1639493ec55c832279c34ce7af037805a23354","observation_id":"75ee56c0-637b-4033-af2f-4716bda53048","resolution":{"observed_at":"2026-07-03T01:27:30.682419Z","resolver_source":"arxiv_id","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:bb88feac1806abdfd602456fa686d0cd2f81931cca82f5cf033a3808ce94be41","observation_id":"b870de9b-894b-492f-9077-6e0e1b349423","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"As introduced in Section 2.3, this interface abstracts raw backend commands into high-level text primitives to form a unified MLLM-native action space","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:37e13de7a7bfb0db90e63d8a97cae1e40f3576046850c3dead4782db5e0bb46b","observation_id":"8cf7a7c7-f8ea-4dde-97ac-e4a64d326be7","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-06-27T16:35:14.099586Z","title":"Step sizes vary by environment: AI2-THOR / Proc- THOR / VirtualHome use Small = 0.25 m, Medium = 0.5 m, Large = 1 m","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks","version":1},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-06-27T16:35:14.099586Z"},"links":{"citing_paper":"/paper/2606.09669"},"observation_digest":"sha256:01b068c39daf7231fb223c32bbd9ddbc79244cd06943ac67899ff841019f5e70","observation_id":"1a4b8ee3-f5a1-47ee-8a26-aaa2958fef65","resolution":{"observed_at":"2026-06-27T16:35:14.099586Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2606.09669","last_updated":"2026-06-08T15:51:51Z","latest_version":1,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-02T20:17:41.727375Z","submitted_at":"2026-06-08T15:51:51Z","title":"SpatialWorld: Benchmarking Interactive Spatial Reasoning of Multimodal Agents in Real-World Tasks"},"reference_resolution":{"displayed":94,"state_counts":{"malformed_identifier":1,"metadata_mismatch":2,"parse_uncertain":0,"unresolved":48,"verified_exact":43,"verified_fuzzy":0},"total_outbound_references":94},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 94 of 94 outbound references and 1 inbound Pith citation observation for arXiv:2606.09669."}