{"as_of":"2026-08-03T22:49:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:4cb9e4302af174995073d4e4bca878da43a53c3c0e021a354d13b09b3b8fd889","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":40,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":40,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-03T06:30:56.289259+00:00","state":"measured"},{"denominator":40,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":40,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-03T13:46:28.179433Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-08T02:44:27.702116Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2501.01428","last_updated":"2025-03-11T07:54:04Z","snapshot_observed_at":"2026-07-06T20:15:54.645357Z","submitted_at":"2025-01-02T18:59:59Z","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","version":4},"cited_work":{"arxiv_id":"2501.01428","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.01428","snapshot_observed_at":"2026-07-08T02:44:27.702116Z","title":"Gpt4scene: Understand 3d scenes from videos with vision-language models","venue":"cs.CV","work_id":"1b0a4b52-5390-4922-abed-d044b52ad3bb","year":2025},"citing_paper":{"arxiv_id":"2505.23747","last_updated":"2026-05-19T02:23:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-29T17:59:04Z","title":"Spatial-MLLM: Boosting MLLM Capabilities in Visual-based Spatial Intelligence","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-16T08:34:36.824053Z"},"links":{"cited_paper":"/paper/2501.01428","citing_paper":"/paper/2505.23747"},"observation_digest":"sha256:29dd7a61e7302ef1ac8c8945a5ef329d256ac98a2e8b12ea43fc0a06581d557e","observation_id":"68282466-6117-4a8f-bee3-501a1f557853","resolution":{"observed_at":"2026-05-16T08:34:36.929625Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01428","last_updated":"2025-03-11T07:54:04Z","snapshot_observed_at":"2026-07-06T20:15:54.645357Z","submitted_at":"2025-01-02T18:59:59Z","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","version":4},"cited_work":{"arxiv_id":"2501.01428","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.01428","snapshot_observed_at":"2026-07-08T02:44:27.702116Z","title":"Gpt4scene: Understand 3d scenes from videos with vision-language models","venue":"cs.CV","work_id":"1b0a4b52-5390-4922-abed-d044b52ad3bb","year":2025},"citing_paper":{"arxiv_id":"2505.23747","last_updated":"2026-05-19T02:23:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-29T17:59:04Z","title":"Spatial-MLLM: Boosting MLLM Capabilities in Visual-based Spatial Intelligence","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-22T00:59:13.826054Z"},"links":{"cited_paper":"/paper/2501.01428","citing_paper":"/paper/2505.23747"},"observation_digest":"sha256:fb19bf83fd55a6c5e2e2875cfd598b9b120636dfbc8200e95cfd8d22e3771572","observation_id":"65e82996-56b2-4918-875e-caced0b314df","resolution":{"observed_at":"2026-05-22T01:00:51.395130Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01428","last_updated":"2025-03-11T07:54:04Z","snapshot_observed_at":"2026-07-06T20:15:54.645357Z","submitted_at":"2025-01-02T18:59:59Z","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","version":4},"cited_work":{"arxiv_id":"2501.01428","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.01428","snapshot_observed_at":"2026-07-08T02:44:27.702116Z","title":"Gpt4scene: Understand 3d scenes from videos with vision-language models","venue":"cs.CV","work_id":"1b0a4b52-5390-4922-abed-d044b52ad3bb","year":2025},"citing_paper":{"arxiv_id":"2506.09965","last_updated":"2025-06-19T03:46:55Z","snapshot_observed_at":"2026-07-31T21:40:49.363128Z","submitted_at":"2025-06-11T17:41:50Z","title":"Reinforcing Spatial Reasoning in Vision-Language Models with Interwoven Thinking and Visual Drawing","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-17T04:58:10.202784Z"},"links":{"cited_paper":"/paper/2501.01428","citing_paper":"/paper/2506.09965"},"observation_digest":"sha256:45319d8650d8a42a8355428e3a8966964e0d0e971d5a8cd02640442d7f828588","observation_id":"c1758b5f-e01e-4635-adf9-ceae67bd6f17","resolution":{"observed_at":"2026-05-17T04:58:10.236698Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01428","last_updated":"2025-03-11T07:54:04Z","snapshot_observed_at":"2026-07-06T20:15:54.645357Z","submitted_at":"2025-01-02T18:59:59Z","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","version":4},"cited_work":{"arxiv_id":"2501.01428","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.01428","snapshot_observed_at":"2026-07-08T02:44:27.702116Z","title":"Gpt4scene: Understand 3d scenes from videos with vision-language models","venue":"cs.CV","work_id":"1b0a4b52-5390-4922-abed-d044b52ad3bb","year":2025},"citing_paper":{"arxiv_id":"2511.21471","last_updated":"2026-05-07T07:59:46Z","snapshot_observed_at":"2026-08-02T23:27:20.204280Z","submitted_at":"2025-11-26T15:04:18Z","title":"SpatialBench: Benchmarking Multimodal Large Language Models for Spatial Cognition","version":4},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-17T04:54:59.903644Z"},"links":{"cited_paper":"/paper/2501.01428","citing_paper":"/paper/2511.21471"},"observation_digest":"sha256:659b3b28fd8ef4b2006577f56660c9fa4daad6b663a1b01698ca088f5b89c055","observation_id":"b916cc03-05e3-4628-a445-d4d239b6680e","resolution":{"observed_at":"2026-05-17T04:59:04.210840Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01428","last_updated":"2025-03-11T07:54:04Z","snapshot_observed_at":"2026-07-06T20:15:54.645357Z","submitted_at":"2025-01-02T18:59:59Z","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.01428","snapshot_observed_at":"2026-08-03T13:46:28.179433Z","title":"Gpt4scene: Understand 3d scenes from videos with vision-language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2512.23020","last_updated":"2026-07-07T09:58:55Z","snapshot_observed_at":"2026-08-03T13:46:24.013267Z","submitted_at":"2025-12-28T17:44:20Z","title":"OpenGround: Planning-based Online Perception for Open-World 3D Visual Grounding","version":3},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-03T13:46:28.179433Z"},"links":{"cited_paper":"/paper/2501.01428","citing_paper":"/paper/2512.23020"},"observation_digest":"sha256:9b2ebb941f6b68c190f93b75d687f8af6f4d4e9a1ee177508ad8b798eda8e68c","observation_id":"e6481afa-a945-489a-9415-830dd543b5bd","resolution":{"observed_at":"2026-08-03T13:46:28.179433Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01428","last_updated":"2025-03-11T07:54:04Z","snapshot_observed_at":"2026-07-06T20:15:54.645357Z","submitted_at":"2025-01-02T18:59:59Z","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.01428","snapshot_observed_at":"2026-08-02T22:08:40.003799Z","title":"GPT4Scene: Understand 3D scenes from videos with vision-language models.arXiv preprint arXiv:2501.01428,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2602.18527","last_updated":"2026-05-28T12:11:44Z","snapshot_observed_at":"2026-08-02T22:08:37.805309Z","submitted_at":"2026-02-20T04:06:07Z","title":"JAEGER: Joint 3D Audio-Visual Grounding and Reasoning in Simulated Physical Environments","version":3},"reference_index":2015,"source":"pdf_text","source_observed_at":"2026-08-02T22:08:40.003799Z"},"links":{"cited_paper":"/paper/2501.01428","citing_paper":"/paper/2602.18527"},"observation_digest":"sha256:30aff8056c1cecbd80c4f176786ecb09b114d2eab01ba107521f1164b63a1ec6","observation_id":"48b73ff5-a861-434c-9f81-a910cb1ca327","resolution":{"observed_at":"2026-08-02T22:08:40.003799Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01428","last_updated":"2025-03-11T07:54:04Z","snapshot_observed_at":"2026-07-06T20:15:54.645357Z","submitted_at":"2025-01-02T18:59:59Z","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","version":4},"cited_work":{"arxiv_id":"2501.01428","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.01428","snapshot_observed_at":"2026-07-08T02:44:27.702116Z","title":"Gpt4scene: Understand 3d scenes from videos with vision-language models","venue":"cs.CV","work_id":"1b0a4b52-5390-4922-abed-d044b52ad3bb","year":2025},"citing_paper":{"arxiv_id":"2603.08592","last_updated":"2026-04-26T17:36:17Z","snapshot_observed_at":"2026-07-31T15:55:15.225728Z","submitted_at":"2026-03-09T16:42:43Z","title":"Boosting MLLM Spatial Reasoning with Geometrically Referenced 3D Scene Representations","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-15T14:31:03.909336Z"},"links":{"cited_paper":"/paper/2501.01428","citing_paper":"/paper/2603.08592"},"observation_digest":"sha256:63ca7aaefae5f67ce618a9ea21ffb0d96925e9e86fd9a9254a25242574eb826d","observation_id":"67e5a825-e443-4632-9d14-e9c0fcec8480","resolution":{"observed_at":"2026-05-15T14:35:56.059324Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01428","last_updated":"2025-03-11T07:54:04Z","snapshot_observed_at":"2026-07-06T20:15:54.645357Z","submitted_at":"2025-01-02T18:59:59Z","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.01428","snapshot_observed_at":"2026-08-02T18:06:09.926358Z","title":"arXiv preprint arXiv:2501.01428 (2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.16461","last_updated":"2026-07-20T02:21:38Z","snapshot_observed_at":"2026-08-02T18:06:08.475509Z","submitted_at":"2026-03-17T12:43:48Z","title":"GAP-MLLM: Geometry-Aligned Pre-training for Activating 3D Spatial Perception in Multimodal Large Language Models","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-02T18:06:09.926358Z"},"links":{"cited_paper":"/paper/2501.01428","citing_paper":"/paper/2603.16461"},"observation_digest":"sha256:b4932b2297bc66e32634b54fc91b3463ee811d07471f49992d1f278c68f4a66d","observation_id":"0ee15ea0-69cc-44c5-9b2c-8fcee3ae62f6","resolution":{"observed_at":"2026-08-02T18:06:09.926358Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01428","last_updated":"2025-03-11T07:54:04Z","snapshot_observed_at":"2026-07-06T20:15:54.645357Z","submitted_at":"2025-01-02T18:59:59Z","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","version":4},"cited_work":{"arxiv_id":"2501.01428","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.01428","snapshot_observed_at":"2026-07-08T02:44:27.702116Z","title":"Gpt4scene: Understand 3d scenes from videos with vision-language models","venue":"cs.CV","work_id":"1b0a4b52-5390-4922-abed-d044b52ad3bb","year":2025},"citing_paper":{"arxiv_id":"2603.17980","last_updated":"2026-05-07T05:29:31Z","snapshot_observed_at":"2026-07-06T22:49:37.944352Z","submitted_at":"2026-03-18T17:42:49Z","title":"Feeling the Space: Egomotion-Aware Video Representation for Efficient and Accurate 3D Scene Understanding","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-15T09:30:03.668178Z"},"links":{"cited_paper":"/paper/2501.01428","citing_paper":"/paper/2603.17980"},"observation_digest":"sha256:f60a175d8146a758a0ccbdabdae615379be2cbedad538701cc47ba6e1b951cf1","observation_id":"4b25ab6a-9274-48e0-864f-28bfc1251243","resolution":{"observed_at":"2026-05-15T09:30:22.387360Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01428","last_updated":"2025-03-11T07:54:04Z","snapshot_observed_at":"2026-07-06T20:15:54.645357Z","submitted_at":"2025-01-02T18:59:59Z","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","version":4},"cited_work":{"arxiv_id":"2501.01428","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.01428","snapshot_observed_at":"2026-07-08T02:44:27.702116Z","title":"Gpt4scene: Understand 3d scenes from videos with vision-language models","venue":"cs.CV","work_id":"1b0a4b52-5390-4922-abed-d044b52ad3bb","year":2025},"citing_paper":{"arxiv_id":"2604.02689","last_updated":"2026-04-03T03:32:55Z","snapshot_observed_at":"2026-07-06T22:52:02.162758Z","submitted_at":"2026-04-03T03:32:55Z","title":"Efficient3D: A Unified Framework for Adaptive and Debiased Token Reduction in 3D MLLMs","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-05-13T19:45:33.950587Z"},"links":{"cited_paper":"/paper/2501.01428","citing_paper":"/paper/2604.02689"},"observation_digest":"sha256:978fb02a1e95aa23c2973d692025e3f7922cb209cc55e927e6d599a913a7dbe4","observation_id":"fc2db593-5ae4-4870-a65a-2f0695f038cf","resolution":{"observed_at":"2026-05-13T19:48:11.407430Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01428","last_updated":"2025-03-11T07:54:04Z","snapshot_observed_at":"2026-07-06T20:15:54.645357Z","submitted_at":"2025-01-02T18:59:59Z","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","version":4},"cited_work":{"arxiv_id":"2501.01428","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.01428","snapshot_observed_at":"2026-07-08T02:44:27.702116Z","title":"Gpt4scene: Understand 3d scenes from videos with vision-language models","venue":"cs.CV","work_id":"1b0a4b52-5390-4922-abed-d044b52ad3bb","year":2025},"citing_paper":{"arxiv_id":"2604.03296","last_updated":"2026-03-28T00:54:19Z","snapshot_observed_at":"2026-07-06T22:52:29.705312Z","submitted_at":"2026-03-28T00:54:19Z","title":"3D-IDE: 3D Implicit Depth Emergent","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-14T22:34:04.833557Z"},"links":{"cited_paper":"/paper/2501.01428","citing_paper":"/paper/2604.03296"},"observation_digest":"sha256:9524b58c621ba8fbcbaf14effd5f20914f3f1f8a54ae4cc3cbf95005c5f7686f","observation_id":"a248adb3-8099-4504-aa8b-55f56ec5bdd2","resolution":{"observed_at":"2026-05-14T22:38:11.477870Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01428","last_updated":"2025-03-11T07:54:04Z","snapshot_observed_at":"2026-07-06T20:15:54.645357Z","submitted_at":"2025-01-02T18:59:59Z","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","version":4},"cited_work":{"arxiv_id":"2501.01428","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.01428","snapshot_observed_at":"2026-07-08T02:44:27.702116Z","title":"Gpt4scene: Understand 3d scenes from videos with vision-language models","venue":"cs.CV","work_id":"1b0a4b52-5390-4922-abed-d044b52ad3bb","year":2025},"citing_paper":{"arxiv_id":"2604.03318","last_updated":"2026-05-25T15:52:36Z","snapshot_observed_at":"2026-07-13T14:39:49.573224Z","submitted_at":"2026-04-01T15:28:13Z","title":"EgoMind: Activating Spatial Cognition through Linguistic Reasoning in MLLMs","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-13T22:41:09.840792Z"},"links":{"cited_paper":"/paper/2501.01428","citing_paper":"/paper/2604.03318"},"observation_digest":"sha256:06b9d9efd623079bd6caac1d0b4b72d50d4f1927610a25fa9156eb4937a780ca","observation_id":"c0a9d8a8-42c3-4b3c-b50e-a0de98bba6d2","resolution":{"observed_at":"2026-05-13T22:43:22.885945Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01428","last_updated":"2025-03-11T07:54:04Z","snapshot_observed_at":"2026-07-06T20:15:54.645357Z","submitted_at":"2025-01-02T18:59:59Z","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.01428","snapshot_observed_at":"2026-07-13T14:39:55.177552Z","title":"Gpt4scene: Understand 3d scenes from videos with vision-language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2604.03318","last_updated":"2026-05-25T15:52:36Z","snapshot_observed_at":"2026-07-13T14:39:49.573224Z","submitted_at":"2026-04-01T15:28:13Z","title":"EgoMind: Activating Spatial Cognition through Linguistic Reasoning in MLLMs","version":2},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-07-13T14:39:55.177552Z"},"links":{"cited_paper":"/paper/2501.01428","citing_paper":"/paper/2604.03318"},"observation_digest":"sha256:f954d910b399dc8f8f75deed2cf13af00070602190852e3e89425f5f0a050464","observation_id":"e7577ef1-9c43-4f33-97a1-877f78d5de67","resolution":{"observed_at":"2026-07-13T14:39:55.177552Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01428","last_updated":"2025-03-11T07:54:04Z","snapshot_observed_at":"2026-07-06T20:15:54.645357Z","submitted_at":"2025-01-02T18:59:59Z","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","version":4},"cited_work":{"arxiv_id":"2501.01428","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.01428","snapshot_observed_at":"2026-07-08T02:44:27.702116Z","title":"Gpt4scene: Understand 3d scenes from videos with vision-language models","venue":"cs.CV","work_id":"1b0a4b52-5390-4922-abed-d044b52ad3bb","year":2025},"citing_paper":{"arxiv_id":"2604.05695","last_updated":"2026-04-07T10:45:28Z","snapshot_observed_at":"2026-08-03T03:24:51.867211Z","submitted_at":"2026-04-07T10:45:28Z","title":"Let Geometry GUIDE: Layer-wise Unrolling of Geometric Priors in Multimodal LLMs","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-10T19:32:59.381755Z"},"links":{"cited_paper":"/paper/2501.01428","citing_paper":"/paper/2604.05695"},"observation_digest":"sha256:6eb9c31d01a6c7c6f57cd49d5b563a3d9996ba7120ab210225df5a39265660ff","observation_id":"be164b81-94e3-48fe-859e-8a844c2fc348","resolution":{"observed_at":"2026-05-10T22:50:49.009736Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01428","last_updated":"2025-03-11T07:54:04Z","snapshot_observed_at":"2026-07-06T20:15:54.645357Z","submitted_at":"2025-01-02T18:59:59Z","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","version":4},"cited_work":{"arxiv_id":"2501.01428","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.01428","snapshot_observed_at":"2026-07-08T02:44:27.702116Z","title":"Gpt4scene: Understand 3d scenes from videos with vision-language models","venue":"cs.CV","work_id":"1b0a4b52-5390-4922-abed-d044b52ad3bb","year":2025},"citing_paper":{"arxiv_id":"2604.18260","last_updated":"2026-04-20T13:33:50Z","snapshot_observed_at":"2026-08-02T12:21:23.079653Z","submitted_at":"2026-04-20T13:33:50Z","title":"Geometry-Guided 3D Visual Token Pruning for Video-Language Models","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-10T05:49:38.346274Z"},"links":{"cited_paper":"/paper/2501.01428","citing_paper":"/paper/2604.18260"},"observation_digest":"sha256:2afc9888c9b9060c78c6011c3d2545ff1a90301ef70719b4e50006896813fde6","observation_id":"c5b05d49-f223-4fa4-9dd5-c4da6aee2aad","resolution":{"observed_at":"2026-05-10T05:51:09.965323Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01428","last_updated":"2025-03-11T07:54:04Z","snapshot_observed_at":"2026-07-06T20:15:54.645357Z","submitted_at":"2025-01-02T18:59:59Z","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","version":4},"cited_work":{"arxiv_id":"2501.01428","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.01428","snapshot_observed_at":"2026-07-08T02:44:27.702116Z","title":"Gpt4scene: Understand 3d scenes from videos with vision-language models","venue":"cs.CV","work_id":"1b0a4b52-5390-4922-abed-d044b52ad3bb","year":2025},"citing_paper":{"arxiv_id":"2605.01736","last_updated":"2026-05-03T06:22:14Z","snapshot_observed_at":"2026-07-06T23:14:52.417213Z","submitted_at":"2026-05-03T06:22:14Z","title":"Multi-Scale Gaussian-Language Map for Zero-shot Embodied Navigation and Reasoning","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-10T15:29:56.514907Z"},"links":{"cited_paper":"/paper/2501.01428","citing_paper":"/paper/2605.01736"},"observation_digest":"sha256:e899e0510e5d14c9024f87045d6ce5165254b52b74bc2a4b49f34684350f3134","observation_id":"190ecbbf-3cc5-431f-a4d6-fb673951d3b9","resolution":{"observed_at":"2026-05-11T10:26:00.789767Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01428","last_updated":"2025-03-11T07:54:04Z","snapshot_observed_at":"2026-07-06T20:15:54.645357Z","submitted_at":"2025-01-02T18:59:59Z","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","version":4},"cited_work":{"arxiv_id":"2501.01428","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.01428","snapshot_observed_at":"2026-07-08T02:44:27.702116Z","title":"Gpt4scene: Understand 3d scenes from videos with vision-language models","venue":"cs.CV","work_id":"1b0a4b52-5390-4922-abed-d044b52ad3bb","year":2025},"citing_paper":{"arxiv_id":"2605.10106","last_updated":"2026-05-11T07:20:09Z","snapshot_observed_at":"2026-07-06T23:22:08.560919Z","submitted_at":"2026-05-11T07:20:09Z","title":"ViSRA: A Video-based Spatial Reasoning Agent for Multi-modal Large Language Models","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-12T04:00:23.681682Z"},"links":{"cited_paper":"/paper/2501.01428","citing_paper":"/paper/2605.10106"},"observation_digest":"sha256:ab9670b25dfb619a3470e8326cf34396d719eff38331715fe1a4a4b3a000c84c","observation_id":"5c25f57d-8586-4dea-a859-da4d1f564f55","resolution":{"observed_at":"2026-05-12T06:46:36.416075Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01428","last_updated":"2025-03-11T07:54:04Z","snapshot_observed_at":"2026-07-06T20:15:54.645357Z","submitted_at":"2025-01-02T18:59:59Z","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","version":4},"cited_work":{"arxiv_id":"2501.01428","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.01428","snapshot_observed_at":"2026-07-08T02:44:27.702116Z","title":"Gpt4scene: Understand 3d scenes from videos with vision-language models","venue":"cs.CV","work_id":"1b0a4b52-5390-4922-abed-d044b52ad3bb","year":2025},"citing_paper":{"arxiv_id":"2605.15876","last_updated":"2026-05-20T09:56:58Z","snapshot_observed_at":"2026-08-02T03:46:43.550240Z","submitted_at":"2026-05-15T11:54:17Z","title":"Unlocking Dense Metric Depth Estimation in VLMs","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-20T19:20:04.468206Z"},"links":{"cited_paper":"/paper/2501.01428","citing_paper":"/paper/2605.15876"},"observation_digest":"sha256:5cbf65e8806e3985665e2f33045a9a87a588679f431838b0c6355544c086ec5d","observation_id":"733a178d-8570-40d1-876e-d01d40dfcd6a","resolution":{"observed_at":"2026-05-20T19:23:40.994480Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01428","last_updated":"2025-03-11T07:54:04Z","snapshot_observed_at":"2026-07-06T20:15:54.645357Z","submitted_at":"2025-01-02T18:59:59Z","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","version":4},"cited_work":{"arxiv_id":"2501.01428","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.01428","snapshot_observed_at":"2026-07-08T02:44:27.702116Z","title":"Gpt4scene: Understand 3d scenes from videos with vision-language models","venue":"cs.CV","work_id":"1b0a4b52-5390-4922-abed-d044b52ad3bb","year":2025},"citing_paper":{"arxiv_id":"2605.15876","last_updated":"2026-05-20T09:56:58Z","snapshot_observed_at":"2026-08-02T03:46:43.550240Z","submitted_at":"2026-05-15T11:54:17Z","title":"Unlocking Dense Metric Depth Estimation in VLMs","version":3},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-21T07:54:52.926995Z"},"links":{"cited_paper":"/paper/2501.01428","citing_paper":"/paper/2605.15876"},"observation_digest":"sha256:c375cd278767d8ebf1947162e5cbfc887af597f9e874dfd959da76b3b0f333a6","observation_id":"fc52b842-e9ba-4cf5-89f1-e34b8519cce7","resolution":{"observed_at":"2026-05-21T07:59:51.097379Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01428","last_updated":"2025-03-11T07:54:04Z","snapshot_observed_at":"2026-07-06T20:15:54.645357Z","submitted_at":"2025-01-02T18:59:59Z","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","version":4},"cited_work":{"arxiv_id":"2501.01428","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.01428","snapshot_observed_at":"2026-07-08T02:44:27.702116Z","title":"Gpt4scene: Understand 3d scenes from videos with vision-language models","venue":"cs.CV","work_id":"1b0a4b52-5390-4922-abed-d044b52ad3bb","year":2025},"citing_paper":{"arxiv_id":"2605.24456","last_updated":"2026-05-26T08:53:56Z","snapshot_observed_at":"2026-07-06T23:34:33.493637Z","submitted_at":"2026-05-23T08:07:45Z","title":"EgoProx: Evaluating MLLMs on Egocentric 3D Proximity Reasoning Across a Cognitive Hierarchy","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-06-30T13:28:43.538541Z"},"links":{"cited_paper":"/paper/2501.01428","citing_paper":"/paper/2605.24456"},"observation_digest":"sha256:6d10056f545ed9e19b982bf639a1de751b15f1da4556335c79085bcada04aaaa","observation_id":"d3284230-31a6-43ff-b81e-d4d6ea12c62e","resolution":{"observed_at":"2026-06-30T13:34:40.494005Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01428","last_updated":"2025-03-11T07:54:04Z","snapshot_observed_at":"2026-07-06T20:15:54.645357Z","submitted_at":"2025-01-02T18:59:59Z","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","version":4},"cited_work":{"arxiv_id":"2501.01428","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.01428","snapshot_observed_at":"2026-07-08T02:44:27.702116Z","title":"Gpt4scene: Understand 3d scenes from videos with vision-language models","venue":"cs.CV","work_id":"1b0a4b52-5390-4922-abed-d044b52ad3bb","year":2025},"citing_paper":{"arxiv_id":"2605.25901","last_updated":"2026-05-25T14:29:04Z","snapshot_observed_at":"2026-07-06T23:35:50.787324Z","submitted_at":"2026-05-25T14:29:04Z","title":"AgentGrounder: Zero-Shot 3D Visual Pointcloud Grounding using Multimodal Language Models","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-06-29T22:47:05.747852Z"},"links":{"cited_paper":"/paper/2501.01428","citing_paper":"/paper/2605.25901"},"observation_digest":"sha256:52dd611026e36cb434e9ce62a3c97938e833b393d60a96593e5dd265ab190af7","observation_id":"10454cfd-1cdb-4d81-94e9-768de310711f","resolution":{"observed_at":"2026-06-29T22:54:01.235024Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01428","last_updated":"2025-03-11T07:54:04Z","snapshot_observed_at":"2026-07-06T20:15:54.645357Z","submitted_at":"2025-01-02T18:59:59Z","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","version":4},"cited_work":{"arxiv_id":"2501.01428","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.01428","snapshot_observed_at":"2026-07-08T02:44:27.702116Z","title":"Gpt4scene: Understand 3d scenes from videos with vision-language models","venue":"cs.CV","work_id":"1b0a4b52-5390-4922-abed-d044b52ad3bb","year":2025},"citing_paper":{"arxiv_id":"2605.28490","last_updated":"2026-05-27T13:45:34Z","snapshot_observed_at":"2026-08-03T02:11:58.369650Z","submitted_at":"2026-05-27T13:45:34Z","title":"SSR3D-LLM: Structured Spatial Reasoning via Latent Steps for Fine-Grained Grounding in Unified 3D-LLMs","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-06-29T13:54:04.992565Z"},"links":{"cited_paper":"/paper/2501.01428","citing_paper":"/paper/2605.28490"},"observation_digest":"sha256:f34658f8965fb297a1bffa87b48eba39196a18e87a2ba7aaf3e074a44616c425","observation_id":"578c5a4d-d994-403e-ba8a-c0ac90ec47c1","resolution":{"observed_at":"2026-06-29T14:03:29.805356Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01428","last_updated":"2025-03-11T07:54:04Z","snapshot_observed_at":"2026-07-06T20:15:54.645357Z","submitted_at":"2025-01-02T18:59:59Z","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","version":4},"cited_work":{"arxiv_id":"2501.01428","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.01428","snapshot_observed_at":"2026-07-08T02:44:27.702116Z","title":"Gpt4scene: Understand 3d scenes from videos with vision-language models","venue":"cs.CV","work_id":"1b0a4b52-5390-4922-abed-d044b52ad3bb","year":2025},"citing_paper":{"arxiv_id":"2605.30231","last_updated":"2026-05-28T17:00:52Z","snapshot_observed_at":"2026-07-06T23:39:34.232661Z","submitted_at":"2026-05-28T17:00:52Z","title":"Beyond 3D VQAs: Injecting 3D Spatial Priors into Vision-Language Models for Enhanced Geometric Reasoning","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-06-29T07:47:52.739735Z"},"links":{"cited_paper":"/paper/2501.01428","citing_paper":"/paper/2605.30231"},"observation_digest":"sha256:ae995dc70f26d405a9df8668215b575bac2a06ddc56ea1109f7dc540dbfc715e","observation_id":"37d34a9c-257d-4de4-b4f5-cb1e30cd20f8","resolution":{"observed_at":"2026-06-29T07:53:13.604835Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01428","last_updated":"2025-03-11T07:54:04Z","snapshot_observed_at":"2026-07-06T20:15:54.645357Z","submitted_at":"2025-01-02T18:59:59Z","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","version":4},"cited_work":{"arxiv_id":"2501.01428","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.01428","snapshot_observed_at":"2026-07-08T02:44:27.702116Z","title":"Gpt4scene: Understand 3d scenes from videos with vision-language models","venue":"cs.CV","work_id":"1b0a4b52-5390-4922-abed-d044b52ad3bb","year":2025},"citing_paper":{"arxiv_id":"2606.00095","last_updated":"2026-05-25T08:53:21Z","snapshot_observed_at":"2026-08-02T11:09:28.452176Z","submitted_at":"2026-05-25T08:53:21Z","title":"Bridging the 2D-3D Gap: A Hierarchical Semantic-Geometric Map for Vision Language Navigation","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-06-29T22:33:32.671550Z"},"links":{"cited_paper":"/paper/2501.01428","citing_paper":"/paper/2606.00095"},"observation_digest":"sha256:42f66acaaab53afa844eacc382edf4cc8ab00ce137898c2b6278437fb0cb5257","observation_id":"dbe0e397-02c2-4bfe-a85e-f3bc4b870a69","resolution":{"observed_at":"2026-06-29T22:34:01.361485Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01428","last_updated":"2025-03-11T07:54:04Z","snapshot_observed_at":"2026-07-06T20:15:54.645357Z","submitted_at":"2025-01-02T18:59:59Z","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","version":4},"cited_work":{"arxiv_id":"2501.01428","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.01428","snapshot_observed_at":"2026-07-08T02:44:27.702116Z","title":"Gpt4scene: Understand 3d scenes from videos with vision-language models","venue":"cs.CV","work_id":"1b0a4b52-5390-4922-abed-d044b52ad3bb","year":2025},"citing_paper":{"arxiv_id":"2606.06891","last_updated":"2026-06-05T04:16:24Z","snapshot_observed_at":"2026-08-02T17:56:35.888517Z","submitted_at":"2026-06-05T04:16:24Z","title":"Stream3D-VLM: Online 3D Spatial Understanding with Incremental Geometry Priors","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-06-27T22:49:02.846428Z"},"links":{"cited_paper":"/paper/2501.01428","citing_paper":"/paper/2606.06891"},"observation_digest":"sha256:33d43f19d4b8caaf82d2717127bf3d1066f1649c3f9156ad419a7117b30b5e81","observation_id":"00859666-c515-43b7-b10d-4520d40da0fd","resolution":{"observed_at":"2026-07-02T16:17:09.633602Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01428","last_updated":"2025-03-11T07:54:04Z","snapshot_observed_at":"2026-07-06T20:15:54.645357Z","submitted_at":"2025-01-02T18:59:59Z","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","version":4},"cited_work":{"arxiv_id":"2501.01428","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.01428","snapshot_observed_at":"2026-07-08T02:44:27.702116Z","title":"Gpt4scene: Understand 3d scenes from videos with vision-language models","venue":"cs.CV","work_id":"1b0a4b52-5390-4922-abed-d044b52ad3bb","year":2025},"citing_paper":{"arxiv_id":"2606.19776","last_updated":"2026-06-18T04:24:28Z","snapshot_observed_at":"2026-08-03T12:16:21.711624Z","submitted_at":"2026-06-18T04:24:28Z","title":"Occ-VLM: Occupancy Grounded Vision Language Model for Indoor Scene Understanding","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-06-26T18:40:20.588652Z"},"links":{"cited_paper":"/paper/2501.01428","citing_paper":"/paper/2606.19776"},"observation_digest":"sha256:f4c3ff8447aa932c97dd24fe2e368ee025b33197da8e749f91416cb195b40ba0","observation_id":"5c320a40-97dd-43ea-91ec-3eb6229a6660","resolution":{"observed_at":"2026-07-04T02:59:25.790387Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01428","last_updated":"2025-03-11T07:54:04Z","snapshot_observed_at":"2026-07-06T20:15:54.645357Z","submitted_at":"2025-01-02T18:59:59Z","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","version":4},"cited_work":{"arxiv_id":"2501.01428","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.01428","snapshot_observed_at":"2026-07-08T02:44:27.702116Z","title":"Gpt4scene: Understand 3d scenes from videos with vision-language models","venue":"cs.CV","work_id":"1b0a4b52-5390-4922-abed-d044b52ad3bb","year":2025},"citing_paper":{"arxiv_id":"2606.19915","last_updated":"2026-06-18T08:09:32Z","snapshot_observed_at":"2026-08-02T20:34:20.314290Z","submitted_at":"2026-06-18T08:09:32Z","title":"SpatialSV: Internalizing Interpretable 3D Spatial Awareness in MLLMs via Task-Oriented Visual Supervision","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-06-26T17:59:21.032539Z"},"links":{"cited_paper":"/paper/2501.01428","citing_paper":"/paper/2606.19915"},"observation_digest":"sha256:75c231f5ed7c5f4b4d3138eeb0648eb57ce8b250173c474225d5def41c150c62","observation_id":"5321faac-0870-41a7-987f-5e556943f58d","resolution":{"observed_at":"2026-07-04T03:29:30.832906Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01428","last_updated":"2025-03-11T07:54:04Z","snapshot_observed_at":"2026-07-06T20:15:54.645357Z","submitted_at":"2025-01-02T18:59:59Z","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","version":4},"cited_work":{"arxiv_id":"2501.01428","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.01428","snapshot_observed_at":"2026-07-08T02:44:27.702116Z","title":"Gpt4scene: Understand 3d scenes from videos with vision-language models","venue":"cs.CV","work_id":"1b0a4b52-5390-4922-abed-d044b52ad3bb","year":2025},"citing_paper":{"arxiv_id":"2606.24649","last_updated":"2026-06-25T01:22:24Z","snapshot_observed_at":"2026-08-02T19:44:07.766639Z","submitted_at":"2026-06-23T14:44:04Z","title":"Agentic Collaborative Cognition for Zero-Shot 3D Understanding","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-06-26T00:22:06.082183Z"},"links":{"cited_paper":"/paper/2501.01428","citing_paper":"/paper/2606.24649"},"observation_digest":"sha256:2e79576d90ff6b6f96e7dc50040e356998f86cbbbdc5d505fe7907575287b08f","observation_id":"5d5a29e3-9795-4ee7-8890-f32df3da4a9b","resolution":{"observed_at":"2026-07-04T16:39:57.795867Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01428","last_updated":"2025-03-11T07:54:04Z","snapshot_observed_at":"2026-07-06T20:15:54.645357Z","submitted_at":"2025-01-02T18:59:59Z","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","version":4},"cited_work":{"arxiv_id":"2501.01428","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.01428","snapshot_observed_at":"2026-07-08T02:44:27.702116Z","title":"Gpt4scene: Understand 3d scenes from videos with vision-language models","venue":"cs.CV","work_id":"1b0a4b52-5390-4922-abed-d044b52ad3bb","year":2025},"citing_paper":{"arxiv_id":"2606.24649","last_updated":"2026-06-25T01:22:24Z","snapshot_observed_at":"2026-08-02T19:44:07.766639Z","submitted_at":"2026-06-23T14:44:04Z","title":"Agentic Collaborative Cognition for Zero-Shot 3D Understanding","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-06-26T05:37:41.407624Z"},"links":{"cited_paper":"/paper/2501.01428","citing_paper":"/paper/2606.24649"},"observation_digest":"sha256:61b333f63d83789d44bbd3f03138941e303de5c21b74c713d2cfeb93d257093d","observation_id":"89812ce1-780f-4e00-a266-22b4f6d2de64","resolution":{"observed_at":"2026-07-04T12:59:52.627994Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01428","last_updated":"2025-03-11T07:54:04Z","snapshot_observed_at":"2026-07-06T20:15:54.645357Z","submitted_at":"2025-01-02T18:59:59Z","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","version":4},"cited_work":{"arxiv_id":"2501.01428","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.01428","snapshot_observed_at":"2026-07-08T02:44:27.702116Z","title":"Gpt4scene: Understand 3d scenes from videos with vision-language models","venue":"cs.CV","work_id":"1b0a4b52-5390-4922-abed-d044b52ad3bb","year":2025},"citing_paper":{"arxiv_id":"2606.28060","last_updated":"2026-06-26T13:08:40Z","snapshot_observed_at":"2026-07-07T00:02:14.441610Z","submitted_at":"2026-06-26T13:08:40Z","title":"ReScene: Structured Indoor Scene Reconstruction from Multi-View Captures","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-29T04:30:56.107507Z"},"links":{"cited_paper":"/paper/2501.01428","citing_paper":"/paper/2606.28060"},"observation_digest":"sha256:47db649759e740e0b715bea06bbfef460566f4a7bf29924565b4bbc172fa3887","observation_id":"3c3a0780-16a3-44a3-ab22-b1bf0f143544","resolution":{"observed_at":"2026-06-29T20:03:57.264648Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01428","last_updated":"2025-03-11T07:54:04Z","snapshot_observed_at":"2026-07-06T20:15:54.645357Z","submitted_at":"2025-01-02T18:59:59Z","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","version":4},"cited_work":{"arxiv_id":"2501.01428","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.01428","snapshot_observed_at":"2026-07-08T02:44:27.702116Z","title":"Gpt4scene: Understand 3d scenes from videos with vision-language models","venue":"cs.CV","work_id":"1b0a4b52-5390-4922-abed-d044b52ad3bb","year":2025},"citing_paper":{"arxiv_id":"2607.01784","last_updated":"2026-07-02T06:56:29Z","snapshot_observed_at":"2026-08-03T11:23:56.400674Z","submitted_at":"2026-07-02T06:56:29Z","title":"SpaceEra++: A Unified Framework Towards 3D Spatial Reasoning in Video","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-07-03T16:16:41.412451Z"},"links":{"cited_paper":"/paper/2501.01428","citing_paper":"/paper/2607.01784"},"observation_digest":"sha256:41efc6d622a5eaf3d681fe4c627d22b9786d0238966970f8880893905fada3c0","observation_id":"3d876ced-b9fb-4c34-9a35-ee6c351c74ff","resolution":{"observed_at":"2026-07-03T16:18:37.281330Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01428","last_updated":"2025-03-11T07:54:04Z","snapshot_observed_at":"2026-07-06T20:15:54.645357Z","submitted_at":"2025-01-02T18:59:59Z","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.01428","snapshot_observed_at":"2026-07-12T06:12:48.722467Z","title":"arXiv preprint arXiv:2501.01428 (2025) 5","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.02908","last_updated":"2026-07-03T03:14:17Z","snapshot_observed_at":"2026-07-12T06:12:48.090334Z","submitted_at":"2026-07-03T03:14:17Z","title":"Holo-Captioning: Toward the Text Equivalent of 3D Scenes","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-07-12T06:12:48.722467Z"},"links":{"cited_paper":"/paper/2501.01428","citing_paper":"/paper/2607.02908"},"observation_digest":"sha256:e2aca127717b6cbd8a8304872208af59b17381af3b9c7a394f66216ed0e36f79","observation_id":"94eb3c21-e4a1-4a59-9fde-60aeeb1b28de","resolution":{"observed_at":"2026-07-12T06:12:48.722467Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01428","last_updated":"2025-03-11T07:54:04Z","snapshot_observed_at":"2026-07-06T20:15:54.645357Z","submitted_at":"2025-01-02T18:59:59Z","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.01428","snapshot_observed_at":"2026-07-11T21:50:49.625343Z","title":"Gpt4scene: Un- derstand 3d scenes from videos with vision-language models.arXiv preprint arXiv:2501.01428,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.04079","last_updated":"2026-07-05T02:00:54Z","snapshot_observed_at":"2026-07-11T21:50:49.150051Z","submitted_at":"2026-07-05T02:00:54Z","title":"Seeing Once is Enough? Online Geometry-Aware Token Pruning for 3D Question Answering","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-07-11T21:50:49.625343Z"},"links":{"cited_paper":"/paper/2501.01428","citing_paper":"/paper/2607.04079"},"observation_digest":"sha256:6581babc2cc7c2e116580aaf7fd47f8ad1bd950f778620ed1c6b347137f0df63","observation_id":"feaaff6a-f67b-4390-aa46-1d8ffe0c02de","resolution":{"observed_at":"2026-07-11T21:50:49.625343Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01428","last_updated":"2025-03-11T07:54:04Z","snapshot_observed_at":"2026-07-06T20:15:54.645357Z","submitted_at":"2025-01-02T18:59:59Z","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.01428","snapshot_observed_at":"2026-07-11T19:16:57.396710Z","title":"Gpt4scene: Understand 3d scenes from videos with vision-language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.04426","last_updated":"2026-07-05T17:43:06Z","snapshot_observed_at":"2026-08-02T16:12:54.212857Z","submitted_at":"2026-07-05T17:43:06Z","title":"ACE-Brain-0.5: A Unified Embodied Foundational Model for Physical Agentic AI","version":1},"reference_index":158,"source":"pdf_text","source_observed_at":"2026-07-11T19:16:57.396710Z"},"links":{"cited_paper":"/paper/2501.01428","citing_paper":"/paper/2607.04426"},"observation_digest":"sha256:6456aa1450c686e2775dce4d6c5ed7d723f581080162a60e1e6ce2d399f4f16d","observation_id":"8973bd45-e242-4a6f-ade2-2af26347c810","resolution":{"observed_at":"2026-07-11T19:16:57.396710Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01428","last_updated":"2025-03-11T07:54:04Z","snapshot_observed_at":"2026-07-06T20:15:54.645357Z","submitted_at":"2025-01-02T18:59:59Z","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","version":4},"cited_work":{"arxiv_id":"2501.01428","doi":null,"metadata_source":"pith","pith_arxiv_id":"2501.01428","snapshot_observed_at":"2026-07-08T02:44:27.702116Z","title":"Gpt4scene: Understand 3d scenes from videos with vision-language models","venue":"cs.CV","work_id":"1b0a4b52-5390-4922-abed-d044b52ad3bb","year":2025},"citing_paper":{"arxiv_id":"2607.06534","last_updated":"2026-07-13T03:52:19Z","snapshot_observed_at":"2026-07-16T23:18:47.677814Z","submitted_at":"2026-07-07T17:39:41Z","title":"CAIRN: Cross-Room 3D Scene Understanding with Topology-Aware Large Multimodal Models","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-07-08T02:44:00.608590Z"},"links":{"cited_paper":"/paper/2501.01428","citing_paper":"/paper/2607.06534"},"observation_digest":"sha256:a5b45b940272fb59097bcb3608bc209e527d82b13d1f1d02b96ff7e4535a8f33","observation_id":"b80da959-fcc9-43a8-a6e5-740fb519ec4f","resolution":{"observed_at":"2026-07-08T02:44:27.703386Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-03T06:30:56.289259+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01428","last_updated":"2025-03-11T07:54:04Z","snapshot_observed_at":"2026-07-06T20:15:54.645357Z","submitted_at":"2025-01-02T18:59:59Z","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.01428","snapshot_observed_at":"2026-07-14T16:00:13.133298Z","title":"Gpt4scene: Understand 3d scenes from videos with vision-language models.arXiv preprint arXiv:2501.01428, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.06534","last_updated":"2026-07-13T03:52:19Z","snapshot_observed_at":"2026-07-16T23:18:47.677814Z","submitted_at":"2026-07-07T17:39:41Z","title":"CAIRN: Cross-Room 3D Scene Understanding with Topology-Aware Large Multimodal Models","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-07-14T16:00:13.133298Z"},"links":{"cited_paper":"/paper/2501.01428","citing_paper":"/paper/2607.06534"},"observation_digest":"sha256:19faeee7bf5d0fb2b7d52c3a04ce64ad5e18b31b1209e121fca2145bfcae80a5","observation_id":"fbb5f7de-c5d8-41ae-84f1-d333c7925267","resolution":{"observed_at":"2026-07-14T16:00:13.133298Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01428","last_updated":"2025-03-11T07:54:04Z","snapshot_observed_at":"2026-07-06T20:15:54.645357Z","submitted_at":"2025-01-02T18:59:59Z","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.01428","snapshot_observed_at":"2026-08-02T06:33:28.227722Z","title":"Gpt4scene: Understand 3d scenes from videos with vision-language models.arXiv preprint arXiv:2501.01428, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.12477","last_updated":"2026-07-15T01:51:44Z","snapshot_observed_at":"2026-08-02T19:45:12.158299Z","submitted_at":"2026-07-14T08:04:31Z","title":"Self in Space: Benchmarking Self-Awareness and Spatial Cognition in UAV Embodied Intelligence","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-02T06:33:28.227722Z"},"links":{"cited_paper":"/paper/2501.01428","citing_paper":"/paper/2607.12477"},"observation_digest":"sha256:fe8eda37f35fd776e02d710be7ae167732148a71ff8d309a9f196fbda4a83781","observation_id":"f26e0aa9-9241-4119-8386-f578cd8416fa","resolution":{"observed_at":"2026-08-02T06:33:28.227722Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01428","last_updated":"2025-03-11T07:54:04Z","snapshot_observed_at":"2026-07-06T20:15:54.645357Z","submitted_at":"2025-01-02T18:59:59Z","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.01428","snapshot_observed_at":"2026-08-02T00:23:02.432874Z","title":"Gpt4scene: Understand 3d scenes from videos with vision- language models.arXiv preprint arXiv:2501.01428,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.15054","last_updated":"2026-07-16T14:31:16Z","snapshot_observed_at":"2026-08-02T09:38:56.901727Z","submitted_at":"2026-07-16T14:31:16Z","title":"Beyond Single Expert: Harmonizing Diverse Visual Priors in MLLMs for Spatial Understanding","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-02T00:23:02.432874Z"},"links":{"cited_paper":"/paper/2501.01428","citing_paper":"/paper/2607.15054"},"observation_digest":"sha256:d5ad920388b37bfa501b1ffaa34e67349340ce3e862e916e0eb67596acdba42e","observation_id":"d3b06a84-d677-4a0a-9a70-d483a1a880cd","resolution":{"observed_at":"2026-08-02T00:23:02.432874Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01428","last_updated":"2025-03-11T07:54:04Z","snapshot_observed_at":"2026-07-06T20:15:54.645357Z","submitted_at":"2025-01-02T18:59:59Z","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.01428","snapshot_observed_at":"2026-08-01T12:25:57.302362Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.22721","last_updated":"2026-07-21T20:25:50Z","snapshot_observed_at":"2026-08-01T12:25:56.224504Z","submitted_at":"2026-07-21T20:25:50Z","title":"An Interactive Vision Language Platform for Cognitive Remediation in Schizophrenia","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-01T12:25:57.302362Z"},"links":{"cited_paper":"/paper/2501.01428","citing_paper":"/paper/2607.22721"},"observation_digest":"sha256:9977e2506239643150120d5fb6438d4e3e186fa8062b1ab2e16b6c39f54b3c13","observation_id":"fee6c617-29ba-4ae2-be3b-ec4fcdca95e7","resolution":{"observed_at":"2026-08-01T12:25:57.302362Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01428","last_updated":"2025-03-11T07:54:04Z","snapshot_observed_at":"2026-07-06T20:15:54.645357Z","submitted_at":"2025-01-02T18:59:59Z","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.01428","snapshot_observed_at":"2026-07-31T07:16:36.438659Z","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.28442","last_updated":"2026-07-30T16:14:30Z","snapshot_observed_at":"2026-08-03T00:12:07.008223Z","submitted_at":"2026-07-30T16:14:30Z","title":"ViewMind3D: Modular View-Aware Inference for Training-Free 3D-QA","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-07-31T07:16:36.438659Z"},"links":{"cited_paper":"/paper/2501.01428","citing_paper":"/paper/2607.28442"},"observation_digest":"sha256:7fbcef30d322a2c178ba49f8159721348755cc2725a509906a300debf40488fd","observation_id":"4f72c9d9-a190-48ce-8cee-9d7193ae3242","resolution":{"observed_at":"2026-07-31T07:16:36.438659Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2501.01428/citation-record","integrity":"/paper/2501.01428/integrity","json":"/paper/2501.01428/citation-record.json","paper":"/paper/2501.01428"},"outbound":[],"paper":{"arxiv_id":"2501.01428","last_updated":"2025-03-11T07:54:04Z","latest_version":4,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T20:15:54.645357Z","submitted_at":"2025-01-02T18:59:59Z","title":"GPT4Scene: Understand 3D Scenes from Videos with Vision-Language Models"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-03T06:30:56.289259+00:00","source":"crossref"},{"observed_at":"2026-08-03T06:30:50.922721+00:00","source":"retraction_watch"}],"thesis":"As of 3 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 40 inbound Pith citation observations for arXiv:2501.01428."}