{"as_of":"2026-08-13T05:49:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:1b012bb3eac08a749415c5dc08305d6b04a7cace43eb4985fdc032a119b53173","coverage":[{"denominator":49,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":49,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T11:40:25.147462Z","state":"measured"},{"denominator":57,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":57,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-12T06:34:41.77262+00:00","state":"measured"},{"denominator":8,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":8,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-02T06:33:30.837244Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T19:50:10.271272Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"cited_work":{"arxiv_id":"2509.02359","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2509.02359","snapshot_observed_at":"2026-07-04T19:50:10.271272Z","title":"Why do mllms struggle with spatial understanding? a systematic analysis from data to architecture","venue":null,"work_id":"0ff3d2c2-b3b3-41b0-8bb3-bae4d41de054","year":2025},"citing_paper":{"arxiv_id":"2603.03944","last_updated":"2026-04-03T20:11:12Z","snapshot_observed_at":"2026-07-06T22:47:46.409064Z","submitted_at":"2026-03-04T11:09:39Z","title":"SCP: Spatial Causal Prediction in Video","version":2},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-05-15T16:47:44.523606Z"},"links":{"cited_paper":"/paper/2509.02359","citing_paper":"/paper/2603.03944"},"observation_digest":"sha256:97efb25bb2767b63bd4882c39a19514f9bceb384c63b2016ef1480db0952db72","observation_id":"ca2531c2-ce95-4d7d-83cd-3b5390d70603","resolution":{"observed_at":"2026-05-15T16:50:11.305915Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"cited_work":{"arxiv_id":"2509.02359","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2509.02359","snapshot_observed_at":"2026-07-04T19:50:10.271272Z","title":"Why do mllms struggle with spatial understanding? a systematic analysis from data to architecture","venue":null,"work_id":"0ff3d2c2-b3b3-41b0-8bb3-bae4d41de054","year":2025},"citing_paper":{"arxiv_id":"2604.13321","last_updated":"2026-04-14T21:57:58Z","snapshot_observed_at":"2026-08-11T20:04:12.538612Z","submitted_at":"2026-04-14T21:57:58Z","title":"Why MLLMs Struggle to Determine Object Orientations","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-10T15:19:44.979076Z"},"links":{"cited_paper":"/paper/2509.02359","citing_paper":"/paper/2604.13321"},"observation_digest":"sha256:de91f9af3ac2b5d6e6bf058ab41e07232fb40487f8787c4f891cffc45f2ab0fd","observation_id":"f8784c33-9dd8-429b-956d-07c8ea5c2d13","resolution":{"observed_at":"2026-05-11T10:46:04.597092Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"cited_work":{"arxiv_id":"2509.02359","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2509.02359","snapshot_observed_at":"2026-07-04T19:50:10.271272Z","title":"Why do mllms struggle with spatial understanding? a systematic analysis from data to architecture","venue":null,"work_id":"0ff3d2c2-b3b3-41b0-8bb3-bae4d41de054","year":2025},"citing_paper":{"arxiv_id":"2605.22100","last_updated":"2026-05-28T08:19:59Z","snapshot_observed_at":"2026-08-12T03:34:58.728529Z","submitted_at":"2026-05-21T07:36:41Z","title":"MPDocBench-Parse: Benchmarking Practical Multi-page Document Parsing","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-05-22T05:58:04.055855Z"},"links":{"cited_paper":"/paper/2509.02359","citing_paper":"/paper/2605.22100"},"observation_digest":"sha256:ac243f92297aa04497231fb8db2bdbe3f440c1bbedeb1b7c5ccf5c4aa651d717","observation_id":"6be88569-d713-4f35-b426-8593bf50f9a3","resolution":{"observed_at":"2026-05-22T06:01:08.940684Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"cited_work":{"arxiv_id":"2509.02359","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2509.02359","snapshot_observed_at":"2026-07-04T19:50:10.271272Z","title":"Why do mllms struggle with spatial understanding? a systematic analysis from data to architecture","venue":null,"work_id":"0ff3d2c2-b3b3-41b0-8bb3-bae4d41de054","year":2025},"citing_paper":{"arxiv_id":"2605.22100","last_updated":"2026-05-28T08:19:59Z","snapshot_observed_at":"2026-08-12T03:34:58.728529Z","submitted_at":"2026-05-21T07:36:41Z","title":"MPDocBench-Parse: Benchmarking Practical Multi-page Document Parsing","version":2},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-06-30T17:37:33.750306Z"},"links":{"cited_paper":"/paper/2509.02359","citing_paper":"/paper/2605.22100"},"observation_digest":"sha256:2ef1d27c20f258f53bae6d8d22976e9b79d7b1bfc115724d763da087718e34a5","observation_id":"9ee60d47-bac9-4064-951c-b7ec93060950","resolution":{"observed_at":"2026-07-01T15:05:48.241818Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"cited_work":{"arxiv_id":"2509.02359","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2509.02359","snapshot_observed_at":"2026-07-04T19:50:10.271272Z","title":"Why do mllms struggle with spatial understanding? a systematic analysis from data to architecture","venue":null,"work_id":"0ff3d2c2-b3b3-41b0-8bb3-bae4d41de054","year":2025},"citing_paper":{"arxiv_id":"2605.30161","last_updated":"2026-05-28T16:18:01Z","snapshot_observed_at":"2026-07-06T23:39:29.458653Z","submitted_at":"2026-05-28T16:18:01Z","title":"Why Far Looks Up: Probing Spatial Representation in Vision-Language Models","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-06-29T08:06:32.403727Z"},"links":{"cited_paper":"/paper/2509.02359","citing_paper":"/paper/2605.30161"},"observation_digest":"sha256:627793cebcd0927c3d25fbd55850ca177d5f2ef56289ce91e846ea2ee573c71d","observation_id":"aa78aa4f-a42b-476b-b5e6-c66dfc775739","resolution":{"observed_at":"2026-06-29T08:13:15.654177Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"cited_work":{"arxiv_id":"2509.02359","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2509.02359","snapshot_observed_at":"2026-07-04T19:50:10.271272Z","title":"Why do mllms struggle with spatial understanding? a systematic analysis from data to architecture","venue":null,"work_id":"0ff3d2c2-b3b3-41b0-8bb3-bae4d41de054","year":2025},"citing_paper":{"arxiv_id":"2606.11770","last_updated":"2026-06-10T07:54:42Z","snapshot_observed_at":"2026-07-06T23:50:49.040880Z","submitted_at":"2026-06-10T07:54:42Z","title":"SVoT: State-aware Visualization-of-Thought for Spatial Reasoning via Reinforcement Learning","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-27T09:59:02.899488Z"},"links":{"cited_paper":"/paper/2509.02359","citing_paper":"/paper/2606.11770"},"observation_digest":"sha256:cbf7bc9105c76d59461931e570f1379113b23994ee9b0751df22c855a79f3b67","observation_id":"ed6f41d5-a745-45da-95ca-f5c128022165","resolution":{"observed_at":"2026-07-03T10:27:56.932385Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"cited_work":{"arxiv_id":"2509.02359","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2509.02359","snapshot_observed_at":"2026-07-04T19:50:10.271272Z","title":"Why do mllms struggle with spatial understanding? a systematic analysis from data to architecture","venue":null,"work_id":"0ff3d2c2-b3b3-41b0-8bb3-bae4d41de054","year":2025},"citing_paper":{"arxiv_id":"2606.25634","last_updated":"2026-06-24T09:38:27Z","snapshot_observed_at":"2026-07-07T00:00:02.505448Z","submitted_at":"2026-06-24T09:38:27Z","title":"SSMNBench: Diagnosing Image-based Cross-View Human-Object Understanding via Single-View Sufficiency and Multi-View Necessity","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-06-25T21:02:44.441202Z"},"links":{"cited_paper":"/paper/2509.02359","citing_paper":"/paper/2606.25634"},"observation_digest":"sha256:df56694921e498abc17caa1a77c82a6604f4f0645df24f65ec157cbb6e60b55f","observation_id":"c0e47212-e3d1-4316-8fe1-d51a03f0261d","resolution":{"observed_at":"2026-07-04T19:50:10.273001Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2509.02359","snapshot_observed_at":"2026-08-02T06:33:30.837244Z","title":"Why do mllms struggle with spatial understanding? a systematic analysis from data to architecture.arXiv preprint arXiv:2509.02359, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.12477","last_updated":"2026-07-15T01:51:44Z","snapshot_observed_at":"2026-08-09T16:15:33.960649Z","submitted_at":"2026-07-14T08:04:31Z","title":"Self in Space: Benchmarking Self-Awareness and Spatial Cognition in UAV Embodied Intelligence","version":2},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-02T06:33:30.837244Z"},"links":{"cited_paper":"/paper/2509.02359","citing_paper":"/paper/2607.12477"},"observation_digest":"sha256:4827541d95a317b1074ab2f4798f3fb34fc0c510ac42fad7f0ae22502d78157b","observation_id":"800777b3-396f-4f56-848a-6d90e020a154","resolution":{"observed_at":"2026-08-02T06:33:30.837244Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2509.02359/citation-record","integrity":"/paper/2509.02359/integrity","json":"/paper/2509.02359/citation-record.json","paper":"/paper/2509.02359"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-05T11:40:20.917876Z","title":"L.; Almeida, D.; Altenschmidt, J.; Altman, S.; Anadkat, S.; et al","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:20.917876Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:90ae9fd660743fca9d64809f2d1d0454835c2687b399e04406a0908d96f57591","observation_id":"d43cbc69-ffeb-4b7c-a279-03935eadac1f","resolution":{"observed_at":"2026-08-05T11:40:20.917876Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:26.180009Z","title":null,"venue":null,"work_id":"f09ef113-cc33-4ad1-945f-d9441bc9875d","year":2022},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:21.002919Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:90c707704026ca6a6d4ba45e6177d7ce2d125aba2d0aef6e46f1a5a2a36a92bf","observation_id":"1fadd94f-a23d-4006-a4ca-231ea67b3297","resolution":{"observed_at":"2026-08-05T11:40:26.184222Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.13923","last_updated":"2025-02-19T18:00:14Z","snapshot_observed_at":"2026-08-12T17:29:41.806995Z","submitted_at":"2025-02-19T18:00:14Z","title":"Qwen2.5-VL Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.13923","snapshot_observed_at":"2026-08-05T11:40:21.062992Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:21.062992Z"},"links":{"cited_paper":"/paper/2502.13923","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:fc04e91b269d060fabb2d6062e024303358b324c7b7139ed6b5ba40fc7eb2d27","observation_id":"c0d61a00-b585-478c-aabb-3d8897018d75","resolution":{"observed_at":"2026-08-05T11:40:21.062992Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:26.165175Z","title":"C.; Geva, M.; He, J.; Wu, J.; and Li, M","venue":null,"work_id":"04eb5202-ec0e-4b7f-b571-cc6cb3c7dbff","year":2025},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:21.124329Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:1e1efa5e46a8bd67c0887fb48623eb8009948b8d250a55236ac45667a957e77f","observation_id":"9fdec59d-0450-4b24-9941-57f9a9bdc608","resolution":{"observed_at":"2026-08-05T11:40:26.169659Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.06192","last_updated":"2024-10-31T18:16:38Z","snapshot_observed_at":"2026-08-12T23:26:30.317326Z","submitted_at":"2024-07-08T17:59:57Z","title":"Multi-Object Hallucination in Vision-Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.06192","snapshot_observed_at":"2026-08-05T11:40:21.185699Z","title":"F.; and Chai, J","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:21.185699Z"},"links":{"cited_paper":"/paper/2407.06192","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:7114e509ac4bf881d4503a6f9982209dc7826f802ef626e5f935597fe5c82422","observation_id":"850362c7-dec3-4fde-8ee5-ebee38b93b0a","resolution":{"observed_at":"2026-08-05T11:40:21.185699Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:26.151489Z","title":null,"venue":null,"work_id":"bd659d37-cc25-4022-97a7-f19af24bb7e0","year":2025},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:21.262546Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:704e163b498fbeb5a04f7b5fc95f9c7d235236f71e84bd141eff24e1036b1fdd","observation_id":"9f49fa7c-3e12-4290-aef7-767d57209960","resolution":{"observed_at":"2026-08-05T11:40:26.155733Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.12391","last_updated":"2025-07-16T16:37:13Z","snapshot_observed_at":"2026-08-12T02:28:25.913425Z","submitted_at":"2025-07-16T16:37:13Z","title":"Assessing the Value of Visual Input: A Benchmark of Multimodal Large Language Models for Robotic Path Planning","version":1},"cited_work":{"arxiv_id":"2507.12391","doi":null,"metadata_source":"pith","pith_arxiv_id":"2507.12391","snapshot_observed_at":"2026-08-05T11:40:25.831314Z","title":"Assessing the Value of Visual Input: A Benchmark of Multimodal Large Language Models for Robotic Path Planning","venue":"cs.RO","work_id":"e6e37e5d-76d1-4673-916d-5b43248e3707","year":2025},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:21.327031Z"},"links":{"cited_paper":"/paper/2507.12391","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:774df3c9b1530ee0ea376bb103566bf408f45bd49efb662ce6b15b40f01765ac","observation_id":"ff00f254-9fce-4018-87cd-dad413f06d7e","resolution":{"observed_at":"2026-08-05T11:40:25.837858Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.06500","last_updated":"2023-06-15T08:00:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-11T00:38:10Z","title":"InstructBLIP: Towards General-purpose Vision-Language Models with Instruction Tuning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.06500","snapshot_observed_at":"2026-08-05T11:40:21.408465Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:21.408465Z"},"links":{"cited_paper":"/paper/2305.06500","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:c278d013fda972a6199b11d73f9d1a160aa150999910d7a2eac3e76c0cb1d526","observation_id":"08d70087-a7c7-41a2-a144-805825c4e51f","resolution":{"observed_at":"2026-08-05T11:40:21.408465Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06788","last_updated":"2025-07-24T10:29:52Z","snapshot_observed_at":"2026-08-09T02:10:24.105469Z","submitted_at":"2025-02-10T18:59:58Z","title":"EVEv2: Improved Baselines for Encoder-Free Vision-Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.06788","snapshot_observed_at":"2026-08-05T11:40:21.474439Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:21.474439Z"},"links":{"cited_paper":"/paper/2502.06788","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:abb4b84111b7ab94bbd826f673964bd09bdfffa2cac0db2d626e4ca0d9ae1800","observation_id":"3c7b937c-c260-4d00-b0f2-9190e4cf32e9","resolution":{"observed_at":"2026-08-05T11:40:21.474439Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11929","last_updated":"2021-06-03T13:08:56Z","snapshot_observed_at":"2026-08-13T02:40:23.887636Z","submitted_at":"2020-10-22T17:55:59Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.11929","snapshot_observed_at":"2026-08-05T11:40:21.539233Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:21.539233Z"},"links":{"cited_paper":"/paper/2010.11929","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:4e8790633b02831d9b9a45bc8a22c3988d330507a11da654046dbf8482db9efe","observation_id":"a7546be4-08d4-4789-aea9-58029f5ab5c9","resolution":{"observed_at":"2026-08-05T11:40:21.539233Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:26.136030Z","title":null,"venue":null,"work_id":"ba6d09ed-0954-4ebe-b37d-c9feeb0ddbb3","year":2024},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:21.609690Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:4aa21e0245282791095cd94a04fb019c813a6672d0d1ccd30415970ea0afae21","observation_id":"163291e9-bbce-4c0d-81d4-877fd88e0107","resolution":{"observed_at":"2026-08-05T11:40:26.141801Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:26.120105Z","title":null,"venue":null,"work_id":"7acc9299-7d5b-47f0-a62e-b19ae4c38644","year":2024},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:21.703116Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:456ead247ad31725d52cd36cdaf1556780c472ab5b5b2fe6550173f8cda00a63","observation_id":"c7ecdfc5-0bc4-4898-84be-61894d17628a","resolution":{"observed_at":"2026-08-05T11:40:26.124659Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.21776","last_updated":"2025-10-22T16:42:24Z","snapshot_observed_at":"2026-08-05T07:15:29.998948Z","submitted_at":"2025-03-27T17:59:51Z","title":"Video-R1: Reinforcing Video Reasoning in MLLMs","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.21776","snapshot_observed_at":"2026-08-05T11:40:21.757867Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:21.757867Z"},"links":{"cited_paper":"/paper/2503.21776","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:71d26302905aacfea3b6f2e8d114b6eb7a357b492679cdb673215fb6d007cb42","observation_id":"568b43a4-4766-4cfd-8a54-cf81b7105324","resolution":{"observed_at":"2026-08-05T11:40:21.757867Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:26.103909Z","title":null,"venue":null,"work_id":"d75dcb96-d8c9-498c-b84c-d0273cffe6ed","year":2024},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:21.847352Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:e0f874dd65db7ed5b8359e9bc601e75b916d03904621d248be85589ac33533db","observation_id":"c3293966-faaa-4a51-aec1-94670e677bca","resolution":{"observed_at":"2026-08-05T11:40:26.108894Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:26.088365Z","title":"J.; Shen, Y.; Wallis, P.; Allen-Zhu, Z.; Li, Y.; Wang, S.; Wang, L.; Chen, W.; et al","venue":null,"work_id":"60e86211-0aa1-49e6-ba27-aa743b904426","year":2022},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:21.902977Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:58a8e92d86a0b42e256ecca67b6f6f92d6259462137017dee593850b94b68739","observation_id":"1db48a6c-7491-4470-a3f8-6bfa46112aa5","resolution":{"observed_at":"2026-08-05T11:40:26.093096Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21276","last_updated":"2024-10-25T17:43:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-25T17:43:01Z","title":"GPT-4o System Card","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.21276","snapshot_observed_at":"2026-08-05T11:40:21.985438Z","title":"P.; Perelman, A.; Ramesh, A.; Clark, A.; Ostrow, A.; Welihinda, A.; Hayes, A.; Radford, A.; et al","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:21.985438Z"},"links":{"cited_paper":"/paper/2410.21276","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:4971bb896c3441194b9778c143baf2be6ffb1b24bf969f9656b8e2f370762453","observation_id":"f8d1a677-5f44-486a-b010-114fcca076dc","resolution":{"observed_at":"2026-08-05T11:40:21.985438Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:22.040115Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:22.040115Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:ca3cd1e2ef2b3a9a94b3919405a298e9f4f8c8d18a88f98199f687d0a88f149f","observation_id":"fcc756ef-b175-4e56-92b9-42691d5bb884","resolution":{"observed_at":"2026-08-05T11:40:22.040115Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.10167","last_updated":"2025-03-18T00:25:47Z","snapshot_observed_at":"2026-08-07T17:07:43.214031Z","submitted_at":"2025-03-13T08:46:32Z","title":"\"Well, Keep Thinking\": Enhancing LLM Reasoning with Adaptive Injection Decoding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.10167","snapshot_observed_at":"2026-08-05T11:40:22.153841Z","title":"Well, Keep Thinking","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:22.153841Z"},"links":{"cited_paper":"/paper/2503.10167","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:27d7fc584dc0fc82f3e66f9b1de1dd3706a6a90ca00d089d0becf6f743e88f3b","observation_id":"0b96c9d4-86fa-4a0a-a84d-f15b19b42214","resolution":{"observed_at":"2026-08-05T11:40:22.153841Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:26.074297Z","title":null,"venue":null,"work_id":"61bad166-f7d9-4817-8d8a-f61fe5111a16","year":2023},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:22.201294Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:31ea4bf8b1955a793bb1e2f8d3ea9b3a0bcd7a823a3e458b719c54bc1792abe7","observation_id":"9b2f27ef-2772-4c7b-9afd-8c80d6d9b29f","resolution":{"observed_at":"2026-08-05T11:40:26.078596Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1712.05474","last_updated":"2022-08-26T17:12:17Z","snapshot_observed_at":"2026-07-06T06:14:28.435222Z","submitted_at":"2017-12-14T23:17:24Z","title":"AI2-THOR: An Interactive 3D Environment for Visual AI","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1712.05474","snapshot_observed_at":"2026-08-05T11:40:22.265805Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:22.265805Z"},"links":{"cited_paper":"/paper/1712.05474","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:f3cc0c6b7a50cea82eac852834cce1929594c3860855bc6b64d6ec0153dec516","observation_id":"321e73f9-a5cb-4c18-9f98-8102aeccbeeb","resolution":{"observed_at":"2026-08-05T11:40:22.265805Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:26.060269Z","title":"A.; et al","venue":null,"work_id":"1cc82f23-a2fe-4413-936d-077499470399","year":2017},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:22.336424Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:e28f291d2df71283970be799c0aeb99a7b051f46b57b40a94c667e8dd1e46cf5","observation_id":"437e3c59-89c7-4fee-8faa-de9207f17962","resolution":{"observed_at":"2026-08-05T11:40:26.064579Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03326","last_updated":"2024-10-26T16:35:13Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-08-06T17:59:44Z","title":"LLaVA-OneVision: Easy Visual Task Transfer","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.03326","snapshot_observed_at":"2026-08-05T11:40:22.438908Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:22.438908Z"},"links":{"cited_paper":"/paper/2408.03326","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:f4bb1bf9f2dc20a63d722af31fbe810756a3c10bc72fdb540073a41edb8b065a","observation_id":"c70964dc-271f-4a34-9fc1-3a19a1fa9d34","resolution":{"observed_at":"2026-08-05T11:40:22.438908Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:26.045528Z","title":null,"venue":null,"work_id":"7f633fcd-05c4-4387-adfc-c427c5e875ce","year":2025},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:22.541628Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:d56a5810716c72cf83e854e04b049ea61a4fdd9fdcd0d7979aef866e1631d3d6","observation_id":"e2e0e38c-2833-4be1-aa0f-1cecfe54f569","resolution":{"observed_at":"2026-08-05T11:40:26.049526Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.16217","last_updated":"2023-12-24T06:38:11Z","snapshot_observed_at":"2026-08-13T04:54:35.575920Z","submitted_at":"2023-12-24T06:38:11Z","title":"ManipLLM: Embodied Multimodal Large Language Model for Object-Centric Robotic Manipulation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.16217","snapshot_observed_at":"2026-08-05T11:40:22.609304Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:22.609304Z"},"links":{"cited_paper":"/paper/2312.16217","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:9d89652af81335c1b60a55d04b0acac2d8b796e095afc8e5dbe8204d63fd114c","observation_id":"6023ceb1-fe25-4dd7-9265-2df50acdd989","resolution":{"observed_at":"2026-08-05T11:40:22.609304Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.00883","last_updated":"2025-04-14T20:12:57Z","snapshot_observed_at":"2026-08-11T19:16:40.810461Z","submitted_at":"2025-04-01T15:11:11Z","title":"Improved Visual-Spatial Reasoning via R1-Zero-Like Training","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.00883","snapshot_observed_at":"2026-08-05T11:40:22.714354Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:22.714354Z"},"links":{"cited_paper":"/paper/2504.00883","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:5bfdf7880146ac15bb13cb6ddc9a774491bfa8bc6983b31897c5324cf3d7a531","observation_id":"c3a40141-58cc-4308-b02d-db22c41b6f97","resolution":{"observed_at":"2026-08-05T11:40:22.714354Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:22.773906Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:22.773906Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:919490cededcc3782f91ae3c17a0d58aed96cf23a670fcdbe322d14f1538d6cb","observation_id":"1e0205e3-8c82-4a78-993b-45b3ead780f1","resolution":{"observed_at":"2026-08-05T11:40:22.773906Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:22.835023Z","title":null,"venue":null,"work_id":null,"year":2014},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:22.835023Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:2e165c91bef525534309a949f2f3c63ab33999bf4bbd530c11c5220f4aec8ade","observation_id":"ede1b8b4-47f2-4291-b58b-c48bb33fa726","resolution":{"observed_at":"2026-08-05T11:40:22.835023Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:22.893886Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:22.893886Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:49e5b235c01c21dbe57a34600f65f91ce4b07079eb27b766a43e36646ed5a1aa","observation_id":"19b493c7-82ac-4842-9a67-268051338242","resolution":{"observed_at":"2026-08-05T11:40:22.893886Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.08268","last_updated":"2025-02-03T21:47:31Z","snapshot_observed_at":"2026-08-11T03:45:32.889856Z","submitted_at":"2024-02-13T07:47:36Z","title":"World Model on Million-Length Video And Language With Blockwise RingAttention","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.08268","snapshot_observed_at":"2026-08-05T11:40:23.027800Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:23.027800Z"},"links":{"cited_paper":"/paper/2402.08268","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:ab198eb728cafc4d9f083f4996c451dd82ceba3f09aa4acf5c3b7c28b36328b1","observation_id":"6075b8c2-d58a-4fb9-975d-1df9b2df2341","resolution":{"observed_at":"2026-08-05T11:40:23.027800Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.10074","last_updated":"2025-01-23T02:31:25Z","snapshot_observed_at":"2026-08-10T21:32:39.307025Z","submitted_at":"2025-01-17T09:46:27Z","title":"SpatialCoT: Advancing Spatial Reasoning through Coordinate Alignment and Chain-of-Thought for Embodied Task Planning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.10074","snapshot_observed_at":"2026-08-05T11:40:23.101934Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:23.101934Z"},"links":{"cited_paper":"/paper/2501.10074","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:ccdbae132e9ef07508d69520bbf173059b113ed030453591ccb72e1f20038789","observation_id":"7ee2fe4f-6421-460a-b40b-9b26b89109b7","resolution":{"observed_at":"2026-08-05T11:40:23.101934Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:26.010703Z","title":null,"venue":null,"work_id":"36f4d3ee-c045-4ccf-a33a-0ae5aa8eb846","year":2025},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:23.186680Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:ebba30e7c832b70b095c111250713ad6972edbe8684b403a328b1649a93f6e33","observation_id":"3f266c45-5168-4cfd-8d3d-37bd68848f6c","resolution":{"observed_at":"2026-08-05T11:40:26.014987Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.01805","last_updated":"2025-05-21T09:38:44Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-02T15:12:17Z","title":"SpaceR: Reinforcing MLLMs in Video Spatial Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.01805","snapshot_observed_at":"2026-08-05T11:40:23.287038Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:23.287038Z"},"links":{"cited_paper":"/paper/2504.01805","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:4decd318963dd95794faaceb9dc7214c76fd6c67456a497643c4d82792b1a915","observation_id":"cbcd96db-b8ce-418d-b8bd-df236d371313","resolution":{"observed_at":"2026-08-05T11:40:23.287038Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2103.00020","last_updated":"2021-02-26T19:04:58Z","snapshot_observed_at":"2026-07-06T10:45:03.059688Z","submitted_at":"2021-02-26T19:04:58Z","title":"Learning Transferable Visual Models From Natural Language Supervision","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2103.00020","snapshot_observed_at":"2026-08-05T11:40:23.409237Z","title":"W.; Hallacy, C.; Ramesh, A.; Goh, G.; Agarwal, S.; Sastry, G.; Askell, A.; Mishkin, P.; Clark, J.; Krueger, G.; and Sutskever, I","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:23.409237Z"},"links":{"cited_paper":"/paper/2103.00020","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:b0dd6a27035db41ff831ea397d5da5652c0986e45a68c77819eb3ae75fc9c989","observation_id":"de4b4843-70f8-4554-b263-123776f5cb38","resolution":{"observed_at":"2026-08-05T11:40:23.409237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:25.996730Z","title":null,"venue":null,"work_id":"ebea23ff-d73e-4c28-82cf-0a8f4e482006","year":2024},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:23.532237Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:fe8c1b9973a8a4a8c781d9637f657eb0faf38e21c86e981c8425125c23bc50b8","observation_id":"97553819-fd5a-44e4-9949-f61828487942","resolution":{"observed_at":"2026-08-05T11:40:26.000797Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:25.982741Z","title":null,"venue":null,"work_id":"673fdb87-b5ca-432e-b604-6e8cf78760ce","year":2024},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:23.645591Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:616065003c8b1161dccdf2513e592374862686d20fd41113052706fc31448bf9","observation_id":"bfdee345-9c44-4db4-8531-9b1bcd881ded","resolution":{"observed_at":"2026-08-05T11:40:25.987127Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:23.706928Z","title":"V.; Zhou, D.; et al","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:23.706928Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:420dd2216a40784797ddfefb60ace7aab0eb11324220097a2c6adb772654f65c","observation_id":"f8855729-854d-4693-86b5-e93008d60952","resolution":{"observed_at":"2026-08-05T11:40:23.706928Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.05173","last_updated":"2025-05-30T03:54:16Z","snapshot_observed_at":"2026-08-09T16:45:21.957564Z","submitted_at":"2025-02-07T18:56:04Z","title":"VideoRoPE: What Makes for Good Video Rotary Position Embedding?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.05173","snapshot_observed_at":"2026-08-05T11:40:23.835237Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:23.835237Z"},"links":{"cited_paper":"/paper/2502.05173","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:518b434b0b0a48b88a5014bbb19d63bb5aa9915a21c925575a318cee7956cdb2","observation_id":"403e88a9-0bf6-44aa-8100-3224b1e886ed","resolution":{"observed_at":"2026-08-05T11:40:23.835237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.23747","last_updated":"2026-05-19T02:23:16Z","snapshot_observed_at":"2026-08-12T18:05:43.082727Z","submitted_at":"2025-05-29T17:59:04Z","title":"Spatial-MLLM: Boosting MLLM Capabilities in Visual-based Spatial Intelligence","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.23747","snapshot_observed_at":"2026-08-05T11:40:23.949207Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:23.949207Z"},"links":{"cited_paper":"/paper/2505.23747","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:f527d0798e2e46283b87820a4aa81b2bfffd89daef9cda51c76745dc9db95df0","observation_id":"ef897e87-fbc2-49b5-94b3-bc2454b30aaa","resolution":{"observed_at":"2026-08-05T11:40:23.949207Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.12799","last_updated":"2025-03-24T11:30:58Z","snapshot_observed_at":"2026-08-12T01:00:02.203358Z","submitted_at":"2025-03-17T04:07:47Z","title":"Grounded Chain-of-Thought for Multimodal Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.12799","snapshot_observed_at":"2026-08-05T11:40:24.078659Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:24.078659Z"},"links":{"cited_paper":"/paper/2503.12799","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:d2c135cd537dc975b4c2dfd61e9fe3678be5cac4533c415c82a0af1f4c71fa1b","observation_id":"f2774bf3-af99-407d-af29-21f65c5c471d","resolution":{"observed_at":"2026-08-05T11:40:24.078659Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:25.959089Z","title":null,"venue":null,"work_id":"2a9f1c29-2eec-44c5-8e9f-1e448aaaf780","year":2024},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:24.164678Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:43bda6b582d7e4bed17468b534eb0090fd24ff330fa4f168d421b58121e0d6e4","observation_id":"d704a65f-af9c-4d3c-afde-1b14391faacc","resolution":{"observed_at":"2026-08-05T11:40:25.963261Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.12148","last_updated":"2023-12-19T13:31:24Z","snapshot_observed_at":"2026-08-13T04:58:35.298301Z","submitted_at":"2023-12-19T13:31:24Z","title":"Parameter-Efficient Fine-Tuning Methods for Pretrained Language Models: A Critical Review and Assessment","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.12148","snapshot_observed_at":"2026-08-05T11:40:24.298029Z","title":"J.; Tao, X.; and Wang, F","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:24.298029Z"},"links":{"cited_paper":"/paper/2312.12148","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:bc879722d56ce796a01299e1b3660d60ef8293aa665a862b218d43629ca8f693","observation_id":"b9820517-4e69-4ed2-8b76-c3d73ac9ccbe","resolution":{"observed_at":"2026-08-05T11:40:24.298029Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.14171","last_updated":"2025-07-02T21:00:36Z","snapshot_observed_at":"2026-08-10T08:02:13.965614Z","submitted_at":"2024-12-18T18:59:54Z","title":"Thinking in Space: How Multimodal Large Language Models See, Remember, and Recall Spaces","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.14171","snapshot_observed_at":"2026-08-05T11:40:24.437770Z","title":"W.; Han, R.; Fei-Fei, L.; and Xie, S","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:24.437770Z"},"links":{"cited_paper":"/paper/2412.14171","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:5182208bec24eca550865114f4c299a25fa38cf369d5e9ee649d91f1b50d9fff","observation_id":"af90493a-917d-49dd-990a-54b38a3e4d24","resolution":{"observed_at":"2026-08-05T11:40:24.437770Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:25.944909Z","title":null,"venue":null,"work_id":"f8e5a260-822f-41fe-983e-003f908d0042","year":2025},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:24.532215Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:6fdf966188fbc7f9c91fb669ccb175e92c091f14468cfea3bec06dd0db8f743e","observation_id":"5b34fe1e-369b-4e4f-8a31-1f4ff23b8b7d","resolution":{"observed_at":"2026-08-05T11:40:25.949071Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:25.930335Z","title":null,"venue":null,"work_id":"a25112a0-5aa0-410d-abb4-3ac576017e01","year":2025},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:24.634793Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:51ddbea1714a29dd181e7573afba2f862b842b2208a081156e858714d494510f","observation_id":"8956414d-be47-464a-8362-24872587b1fc","resolution":{"observed_at":"2026-08-05T11:40:25.934472Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.00493","last_updated":"2025-03-27T10:30:42Z","snapshot_observed_at":"2026-08-12T23:13:45.948951Z","submitted_at":"2024-11-30T14:28:53Z","title":"Video-3D LLM: Learning Position-Aware Video Representation for 3D Scene Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.00493","snapshot_observed_at":"2026-08-05T11:40:24.704510Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:24.704510Z"},"links":{"cited_paper":"/paper/2412.00493","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:7c8a822490462d93d2124c0aae077ece9fea8a9e956e3524ff4c609ecafcbe77","observation_id":"2d0cd3b5-f8a4-4fb0-95d8-c5b13fa3842e","resolution":{"observed_at":"2026-08-05T11:40:24.704510Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:25.915030Z","title":null,"venue":null,"work_id":"4016c94e-c7ab-49ba-89f3-20afd23b9f06","year":2024},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:24.841237Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:a6dc5e844fba3e9a3a6896e4ba365c3e7fdd7b698675caca621fd2c1ef582bcb","observation_id":"c9dd781a-e214-460d-9585-0f1e0eddc8b5","resolution":{"observed_at":"2026-08-05T11:40:25.919917Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.10479","last_updated":"2025-04-19T03:47:21Z","snapshot_observed_at":"2026-08-10T18:37:57.419939Z","submitted_at":"2025-04-14T17:59:25Z","title":"InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.10479","snapshot_observed_at":"2026-08-05T11:40:24.932938Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:24.932938Z"},"links":{"cited_paper":"/paper/2504.10479","citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:2e47099cf8d0a1b02d6c4db19582603df57cd48c670aa284af584574d3ec1a82","observation_id":"45e98c5a-18a5-468e-8f65-9ca4a5819e59","resolution":{"observed_at":"2026-08-05T11:40:24.932938Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:25.030350Z","title":", \" * write output.state after.block = add.period write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:25.030350Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:ae8c2d96d96100fb0137b10caefcc44c6e336a2a1d4c7d7a007bd81b261f43ab","observation_id":"beec9b9e-010f-42fa-a22b-4ea66791e88f","resolution":{"observed_at":"2026-08-05T11:40:25.030350Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T11:40:25.147462Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-05T11:40:25.147462Z"},"links":{"citing_paper":"/paper/2509.02359"},"observation_digest":"sha256:90573f7f7adc5595d9b4dcb49f4e430e418797862ecc75db0cbfafc1a4985d40","observation_id":"f1de502a-8bf3-41f3-8753-cc32b6943997","resolution":{"observed_at":"2026-08-05T11:40:25.147462Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2509.02359","last_updated":"2025-09-02T14:22:43Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-08T06:17:37.928618Z","submitted_at":"2025-09-02T14:22:43Z","title":"Why Do MLLMs Struggle with Spatial Understanding? A Systematic Analysis from Data to Architecture"},"reference_resolution":{"displayed":49,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":45,"verified_exact":1,"verified_fuzzy":3},"total_outbound_references":49},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"thesis":"As of 13 August 2026, this Paper Citation Record lists 49 of 49 outbound references and 8 inbound Pith citation observations for arXiv:2509.02359."}