{"as_of":"2026-08-06T18:13:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:df24c1ef153f1965996da7cb02c50becb8751c8eaeebbb55b1d372da2ea92623","coverage":[{"denominator":49,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":49,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-10T00:45:46.261005Z","state":"measured"},{"denominator":49,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":49,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2604.20570/citation-record","integrity":"/paper/2604.20570/integrity","json":"/paper/2604.20570/citation-record.json","paper":"/paper/2604.20570"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":"2303.08774","doi":"10.1002/tea.20265","metadata_source":"pith","pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GPT-4 Technical Report","venue":"cs.CL","work_id":"b928e041-6991-4c08-8c81-0359e4097c7b","year":2023},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:9ac73db9ae8f9a45eca589954947b35127c771092a8e9eeb622cd8b0ac36cd50","observation_id":"cda9bf8a-f94c-4cac-b2e3-225874c0af8e","resolution":{"observed_at":"2026-05-10T00:49:48.861002Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2508.13142","doi":"10.48550/arxiv.2508.13142","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Has gpt-5 achieved spatial intelligence? an empirical study","venue":"arXiv (Cornell University)","work_id":"b0d24cae-8b82-4d58-9b26-2a803fd083cc","year":2025},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:6ba52b0760acb24fc1e855fe5ef04c93de7328f1f5bd6b3a49b375870baf64d7","observation_id":"73500b2c-2954-4ffd-ad68-30a4aa3a9da0","resolution":{"observed_at":"2026-05-10T00:49:48.846455Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Revision: Rendering tools enable spatial fidelity in vision-language models","venue":null,"work_id":"4ac0af6a-54a2-4085-b8e6-a22af0ac4bd1","year":2024},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:d13888f22616342f81f256ce2bdf8f9dbfc0ec19ea69cad6471644f46a72600f","observation_id":"a34001a4-7e2e-4b84-b6a0-16ead67f3fc1","resolution":{"observed_at":"2026-05-23T10:42:51.994912Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.12168","last_updated":"2024-01-22T18:01:01Z","snapshot_observed_at":"2026-08-02T19:24:45.895739Z","submitted_at":"2024-01-22T18:01:01Z","title":"SpatialVLM: Endowing Vision-Language Models with Spatial Reasoning Capabilities","version":1},"cited_work":{"arxiv_id":"2401.12168","doi":null,"metadata_source":"pith","pith_arxiv_id":"2401.12168","snapshot_observed_at":"2026-07-10T03:06:43.700595Z","title":"SpatialVLM: Endowing vision-language models with spatial reasoning capabilities","venue":"cs.CV","work_id":"dd74ba70-5068-4b69-ba35-988d213c87bb","year":2024},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"cited_paper":"/paper/2401.12168","citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:251f5c5a37edd306f03f2302a17584a661c588fb1f411af305904bf20f598508","observation_id":"4042bd62-411d-4827-a6a5-d300abfe0de5","resolution":{"observed_at":"2026-05-10T00:49:48.849225Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09568","last_updated":"2025-05-14T17:11:07Z","snapshot_observed_at":"2026-07-06T21:23:57.084147Z","submitted_at":"2025-05-14T17:11:07Z","title":"BLIP3-o: A Family of Fully Open Unified Multimodal Models-Architecture, Training and Dataset","version":1},"cited_work":{"arxiv_id":"2505.09568","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.09568","snapshot_observed_at":"2026-07-08T00:04:22.585543Z","title":"BLIP3-o: A Family of Fully Open Unified Multimodal Models-Architecture, Training and Dataset","venue":"cs.CV","work_id":"86d896d2-592f-4d9b-938e-dfeb11f9388f","year":2025},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"cited_paper":"/paper/2505.09568","citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:edce222ec09adf027767f7abababf23c99528ad6519b34b68568bd939c26db90","observation_id":"cecdd05b-7720-4ad2-a7f9-ca5f984ffe30","resolution":{"observed_at":"2026-05-10T00:49:48.852004Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.17811","last_updated":"2025-01-29T18:00:19Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-29T18:00:19Z","title":"Janus-Pro: Unified Multimodal Understanding and Generation with Data and Model Scaling","version":1},"cited_work":{"arxiv_id":"2501.17811","doi":"10.48550/arxiv.2501.17811","metadata_source":"pith","pith_arxiv_id":"2501.17811","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Janus-Pro: Unified Multimodal Understanding and Generation with Data and Model Scaling","venue":"cs.AI","work_id":"67d9e391-26d1-459e-ab56-07e60511c886","year":2025},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"cited_paper":"/paper/2501.17811","citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:a4d6f91a6b19a57284d3af59f903f612042dd66dd2aeea2891188a5264edb3bf","observation_id":"25fd6349-3bf7-41eb-aadc-d7e5a324e7b9","resolution":{"observed_at":"2026-05-11T08:14:53.293790Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.06261","last_updated":"2025-12-19T14:25:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-07T17:36:04Z","title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","version":6},"cited_work":{"arxiv_id":"2507.06261","doi":"10.48550/arxiv.2503.19","metadata_source":"pith","pith_arxiv_id":"2507.06261","snapshot_observed_at":"2026-07-11T03:17:51.364436Z","title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","venue":"cs.CL","work_id":"008df105-2fdd-45d8-857a-8e35868aecb6","year":2025},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"cited_paper":"/paper/2507.06261","citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:7743589688a0d06dcbf77a441135867ad615c00a041f2af18da33015bb817656","observation_id":"d077e485-6555-40a3-9a7c-97de252b1e5d","resolution":{"observed_at":"2026-05-10T00:49:48.863417Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.26583","last_updated":"2025-10-30T15:11:16Z","snapshot_observed_at":"2026-07-31T17:39:49.561332Z","submitted_at":"2025-10-30T15:11:16Z","title":"Emu3.5: Native Multimodal Models are World Learners","version":1},"cited_work":{"arxiv_id":"2510.26583","doi":"10.48550/arxiv.2510.26583","metadata_source":"pith","pith_arxiv_id":"2510.26583","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Emu3.5: Native Multimodal Models are World Learners","venue":"cs.CV","work_id":"518b061e-87a3-45f9-8419-30a14d89b122","year":2025},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"cited_paper":"/paper/2510.26583","citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:7b55c4ab8a508350ddf62059e6511e7a6e17060ed9b244859bfd6b6d480bd445","observation_id":"a0ee33fa-5819-4e27-96ce-84c4528d6fbe","resolution":{"observed_at":"2026-05-18T01:12:13.931083Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-13T15:49:30.986785+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-13T15:49:30.986785+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Scannet: Richly-annotated 3d reconstructions of indoor scenes","venue":null,"work_id":"162bb0af-afa7-4438-95bf-f243562ba872","year":2017},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:68eccfcce24771377a5fcc961c0e7335c2c311826f36e4186a1ea27b9515ae82","observation_id":"2793333e-7fe9-4de1-bcd0-a66879513da8","resolution":{"observed_at":"2026-05-23T10:42:52.023346Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.14683","last_updated":"2025-07-27T11:45:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-20T17:59:30Z","title":"Emerging Properties in Unified Multimodal Pretraining","version":3},"cited_work":{"arxiv_id":"2505.14683","doi":"10.48550/arxiv.2505.14683","metadata_source":"pith","pith_arxiv_id":"2505.14683","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Emerging Properties in Unified Multimodal Pretraining","venue":"cs.CV","work_id":"e0cfd82c-f5d4-44fd-b531-ec73ab0a805b","year":2025},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"cited_paper":"/paper/2505.14683","citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:562f12c77e474204231ba6c36aad0726790ca4852f75e387ea96e01c3b7e8982","observation_id":"7890a1c9-b446-4591-bb6a-3c0c2f3ba89e","resolution":{"observed_at":"2026-05-10T16:23:42.412235Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1803.10122","last_updated":"2018-05-09T09:06:27Z","snapshot_observed_at":"2026-07-31T21:36:45.596575Z","submitted_at":"2018-03-27T15:08:55Z","title":"World Models","version":4},"cited_work":{"arxiv_id":"1803.10122","doi":"10.48550/arxiv.1712.00409","metadata_source":"pith","pith_arxiv_id":"1803.10122","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"World Models","venue":"cs.LG","work_id":"07227eee-8445-4c98-bce4-c6a6fd5ed907","year":2018},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"cited_paper":"/paper/1803.10122","citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:ca07b2f4112d66d9a7a0663639ad0d93457c7fc39ed9fad9d1cb0db5cafab564","observation_id":"8dee9d7c-4227-43c7-8870-57ee2ee6a2c0","resolution":{"observed_at":"2026-05-11T03:08:36.665942Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2509.22281","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-29T07:13:16.839155Z","title":"Mesatask: Towards task- driven tabletop scene generation via 3d spatial reasoning","venue":null,"work_id":"ba976d56-4f0e-4c5e-914f-ef233754bbd7","year":2025},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:36e7b18b4d49bb9c81054771c9ee2ef6ba665c0718bc60f3db4d27b027d60ba3","observation_id":"f902589c-37d3-4bb9-b1b5-82c9355bebc0","resolution":{"observed_at":"2026-05-10T00:49:48.793696Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.03895","last_updated":"2026-05-11T17:14:31Z","snapshot_observed_at":"2026-07-30T02:16:56.091370Z","submitted_at":"2025-10-04T18:26:55Z","title":"NoTVLA: Semantics-Preserving Robot Adaptation via Narrative Action Interfaces","version":2},"cited_work":{"arxiv_id":"2510.03895","doi":null,"metadata_source":"pith","pith_arxiv_id":"2510.03895","snapshot_observed_at":"2026-07-02T02:36:27.128521Z","title":"Notvla: Narrowing of dense action trajecto- ries for generalizable robot manipulation","venue":"cs.RO","work_id":"0764554a-bb1c-4e06-95c6-90550fc2d6b6","year":2025},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"cited_paper":"/paper/2510.03895","citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:339aacc55f94dbe346d83743cc7906898a2bae423b85703c5a0110078abe23d8","observation_id":"b0b62bb3-2fde-45af-8ecc-383acbaf793d","resolution":{"observed_at":"2026-05-12T03:42:01.811004Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21276","last_updated":"2024-10-25T17:43:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-25T17:43:01Z","title":"GPT-4o System Card","version":1},"cited_work":{"arxiv_id":"2410.21276","doi":"10.1177/15248380231178756","metadata_source":"pith","pith_arxiv_id":"2410.21276","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GPT-4o System Card","venue":"cs.CL","work_id":"f37bf1c7-4964-4e56-9762-d20da8d9009f","year":2024},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"cited_paper":"/paper/2410.21276","citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:a5cce316a6350c2e4b9ac74f4ca6b8c9c85b12ba2476aca6ddb88edf391aeae2","observation_id":"0c034d74-fbb5-4171-8be1-785175155d5e","resolution":{"observed_at":"2026-05-10T00:49:48.788464Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.21257","last_updated":"2025-03-25T05:46:03Z","snapshot_observed_at":"2026-07-06T20:44:31.199648Z","submitted_at":"2025-02-28T17:30:39Z","title":"RoboBrain: A Unified Brain Model for Robotic Manipulation from Abstract to Concrete","version":2},"cited_work":{"arxiv_id":"2502.21257","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.21257","snapshot_observed_at":"2026-07-04T10:39:45.381569Z","title":"Robobrain: A unified brain model for robotic manipulation from abstract to concrete","venue":null,"work_id":"80a52809-7280-487f-875b-4ced1c97fd0f","year":2025},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"cited_paper":"/paper/2502.21257","citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:fd60116a8898d185f292d0e4e3599fc111457c6fa0e8955678e1f49fde78c5bc","observation_id":"5fb424e7-daca-4320-af26-7d603df45e86","resolution":{"observed_at":"2026-05-10T00:49:48.791114Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2506.03135","doi":"10.48550/arxiv.2506.03135","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Omnispatial: Towards comprehensive spatial reasoning benchmark for vision language models","venue":"Open MIND","work_id":"f971b57a-47ff-414d-80f9-50ccac0c8e78","year":2025},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:30aa957d2b3e2c4e7753806474d8a75a0acfa78f8dcd7d4d6dd47ed756c7daa9","observation_id":"e6a29431-880f-4f59-bebb-bc064244eb6f","resolution":{"observed_at":"2026-05-10T00:49:48.796279Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2502.05628","doi":"10.48550/arxiv.2502.05628","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Anyedit: Edit any knowledge encoded in language models","venue":"ArXiv.org","work_id":"72693ca7-a1b8-46ba-9900-49f8a6b52d4b","year":2025},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:368b9c91721339fbcbf29a36ce8c1f38fca4bcb88b7768ad48471b0f59f53bf5","observation_id":"4bb915bc-585b-4a8d-b015-29bbb0407032","resolution":{"observed_at":"2026-05-10T00:49:48.804055Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1712.05474","last_updated":"2022-08-26T17:12:17Z","snapshot_observed_at":"2026-07-06T06:14:28.435222Z","submitted_at":"2017-12-14T23:17:24Z","title":"AI2-THOR: An Interactive 3D Environment for Visual AI","version":4},"cited_work":{"arxiv_id":"1712.05474","doi":"10.48550/arxiv.1712.05474","metadata_source":"pith","pith_arxiv_id":"1712.05474","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"AI2-THOR: An Interactive 3D Environment for Visual AI","venue":"cs.CV","work_id":"9c86ed28-ea70-424c-bd56-34f59dcad861","year":2017},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"cited_paper":"/paper/1712.05474","citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:76d6a858d7b6e8a709d55fd04ba7e3dad631fd7b2deb388d13d9bbc079d87053","observation_id":"ed58cc9f-8363-4bae-9646-4c799ee12a3e","resolution":{"observed_at":"2026-05-12T05:24:31.084211Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.03147","last_updated":"2025-06-18T18:00:05Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-03T17:59:33Z","title":"UniWorld-V1: High-Resolution Semantic Encoders for Unified Visual Understanding and Generation","version":4},"cited_work":{"arxiv_id":"2506.03147","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.03147","snapshot_observed_at":"2026-07-05T16:51:14.197597Z","title":"UniWorld-V1: High-Resolution Semantic Encoders for Unified Visual Understanding and Generation","venue":"cs.CV","work_id":"488a273e-95d8-46f1-87c7-2244068d00d0","year":2025},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"cited_paper":"/paper/2506.03147","citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:99ae72b7a010b7388b9c4794fbc955884b8df4b0b4abb7bf1ac7d54116616f22","observation_id":"37068dec-8091-4623-9883-25b743250ba9","resolution":{"observed_at":"2026-05-12T17:34:27.304691Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Improved baselines with visual instruction tuning","venue":null,"work_id":"2ccc518f-f7f2-43ab-ab03-d577170bf66e","year":2024},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:bb647c0ef834dff35a847f1a66b3e8adbf2e7f90dfd482d441eccb1632922295","observation_id":"2f20d877-967d-4f57-ad1f-7fe0c5547c0d","resolution":{"observed_at":"2026-05-23T10:42:51.958677Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.08268","last_updated":"2025-02-03T21:47:31Z","snapshot_observed_at":"2026-08-05T16:01:06.026395Z","submitted_at":"2024-02-13T07:47:36Z","title":"World Model on Million-Length Video And Language With Blockwise RingAttention","version":4},"cited_work":{"arxiv_id":"2402.08268","doi":null,"metadata_source":"pith","pith_arxiv_id":"2402.08268","snapshot_observed_at":"2026-07-04T10:59:46.566942Z","title":"World Model on Million-Length Video And Language With Blockwise RingAttention","venue":"cs.LG","work_id":"162054d2-6f8e-4dc0-87b8-ebff78210c7c","year":2024},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"cited_paper":"/paper/2402.08268","citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:ecc3204e5162ed4ce81d49e4c5b1fdd4cf4850be9c12f9c446d178763a198413","observation_id":"79c2cc70-f0e2-4419-aefe-a023fdb29698","resolution":{"observed_at":"2026-05-16T06:36:57.375169Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-06T20:02:55.746615Z","title":"Janusflow: Harmonizing autore- gression and rectified flow for unified multimodal under- standing and generation","venue":null,"work_id":"49ac196c-9855-468d-bfd1-018fff53ff2b","year":null},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:b8ebd2878182555aa56add9ed67889e0f45ca188b91d3c7535fa87914d6e1cbb","observation_id":"cdeedbce-6c78-47fa-8442-0b11acc1d89e","resolution":{"observed_at":"2026-05-23T10:42:51.979210Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Membership determination in open clusters using the dbscan clustering algorithm.Astronomy and Computing, 47:100826","venue":null,"work_id":"ea0a9456-0751-4e6e-99a2-07c76bdbdda9","year":2024},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:e23a6adee7907a41223724374603433cf94dab0f95fe414856f26da5055b49ba","observation_id":"7f7d9a11-a775-4956-82b3-e7d7f6c875f2","resolution":{"observed_at":"2026-05-23T10:42:51.990980Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2412.07755","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-05T11:41:02.649286Z","title":"Sat: Spa- tial aptitude training for multimodal language models","venue":null,"work_id":"bbf5f212-8e91-4bac-9e1c-ba1d133bfc4c","year":2024},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:21f93063b8d3f21c2c8c3a1ecfbc92ab4db76ca456adaa1b24695e29de0767ea","observation_id":"b16cfd15-63a4-4686-965c-ee5f0d7461c8","resolution":{"observed_at":"2026-05-10T00:49:48.832912Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Com- mon objects in 3d: Large-scale learning and evaluation of real-life 3d category reconstruction","venue":null,"work_id":"f9b94b71-8b4d-4d59-826a-17137e291f86","year":2021},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:61ff96f795dd597775aa0659c05398adc0c06a581bfbb2f3980a04ea459e4cbc","observation_id":"0835e6c2-6af5-4d3f-bddc-8c65ce6d6c12","resolution":{"observed_at":"2026-05-23T10:42:52.009691Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.23918","last_updated":"2025-07-03T16:49:39Z","snapshot_observed_at":"2026-07-06T21:49:47.325553Z","submitted_at":"2025-06-30T14:48:35Z","title":"Thinking with Images for Multimodal Reasoning: Foundations, Methods, and Future Frontiers","version":3},"cited_work":{"arxiv_id":"2506.23918","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.23918","snapshot_observed_at":"2026-07-11T03:17:51.399835Z","title":"Thinking with Images for Multimodal Reasoning: Foundations, Methods, and Future Frontiers","venue":"cs.CV","work_id":"760ebd7d-977d-4280-afae-adb421d49ed4","year":2025},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"cited_paper":"/paper/2506.23918","citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:b17d2a8b72e0d174cb52050eef41b6c1d1d8a27ecb9a8a8fe3a40299e0f760b5","observation_id":"83c1962e-715b-4ac1-8cd6-d203f4fade51","resolution":{"observed_at":"2026-05-13T08:34:23.540456Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":"2312.11805","doi":"10.1038/nrn2888","metadata_source":"pith","pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gemini: A Family of Highly Capable Multimodal Models","venue":"cs.CL","work_id":"83f7c85b-3f11-450f-ac0c-64d9745220b2","year":2023},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:c2b4ee24f8510b11e241b1de7c7fe7f8e10397915821acfbfc552b00d1333e77","observation_id":"aa9d1fc8-1405-4782-b9a5-2b68bc10d18e","resolution":{"observed_at":"2026-05-10T00:49:48.822406Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12191","last_updated":"2024-10-03T15:54:49Z","snapshot_observed_at":"2026-08-06T05:35:29.109022Z","submitted_at":"2024-09-18T17:59:32Z","title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","version":2},"cited_work":{"arxiv_id":"2409.12191","doi":"10.48550/arxiv.2409.12191","metadata_source":"pith","pith_arxiv_id":"2409.12191","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","venue":"cs.CV","work_id":"8abcfe4f-e0fb-44b7-9123-448fac95f90a","year":2024},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"cited_paper":"/paper/2409.12191","citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:cf513acb034a6d92da6020468d3aa4b125b91bf0431c9410e9365f0b16a2571b","observation_id":"0faa8e75-e042-44d5-a455-6bb80a30d7a7","resolution":{"observed_at":"2026-05-10T00:49:48.811785Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T02:19:33.884263+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T02:19:33.884263+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.18869","last_updated":"2024-09-27T16:06:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-27T16:06:11Z","title":"Emu3: Next-Token Prediction is All You Need","version":1},"cited_work":{"arxiv_id":"2409.18869","doi":"10.48550/arxiv.2409.18869","metadata_source":"pith","pith_arxiv_id":"2409.18869","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Emu3: Next-Token Prediction is All You Need","venue":"cs.CV","work_id":"720d288e-fac0-464c-9929-19efd9a52afc","year":2024},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"cited_paper":"/paper/2409.18869","citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:3a2684a6e4ef05964f79ed98ee9ffe8ae527972a6ee929d9fa457d90b3ae562f","observation_id":"2f1dfbaf-a562-441e-afad-fce9552caf02","resolution":{"observed_at":"2026-05-11T10:56:09.993852Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T03:15:58.044493Z","title":"Image quality assessment: from error visibility to structural similarity.IEEE transactions on image processing, 13(4):600–612","venue":null,"work_id":"40a62567-2f6f-48f2-baaa-612f0568c507","year":2004},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:d529d9368dfc8adce06d4d42fbeba923404ed2c67a7494d606e8dc6ff48214ba","observation_id":"9ecc766e-8abf-49c2-8c5b-f36f5d630f88","resolution":{"observed_at":"2026-05-23T10:42:51.999767Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.24870","last_updated":"2025-06-06T14:51:40Z","snapshot_observed_at":"2026-07-06T21:33:50.814913Z","submitted_at":"2025-05-30T17:59:26Z","title":"GenSpace: Benchmarking Spatially-Aware Image Generation","version":2},"cited_work":{"arxiv_id":"2505.24870","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.24870","snapshot_observed_at":"2026-07-04T13:09:50.420268Z","title":"Genspace: Bench- marking spatially-aware image generation","venue":null,"work_id":"ff323a8a-b684-40e5-9b2f-753d10068972","year":2025},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"cited_paper":"/paper/2505.24870","citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:0348d17c0ac7dedd99d7cb182834776c89c17241705dff9b09af8b680b086cd7","observation_id":"74f31468-8f91-4b76-b3e6-4ac6cdbab895","resolution":{"observed_at":"2026-05-10T00:49:48.814426Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.02324","last_updated":"2025-08-04T11:49:20Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-08-04T11:49:20Z","title":"Qwen-Image Technical Report","version":1},"cited_work":{"arxiv_id":"2508.02324","doi":"10.48550/arxiv.2508.02324","metadata_source":"pith","pith_arxiv_id":"2508.02324","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen-Image Technical Report","venue":"cs.CV","work_id":"d06d7ecc-7579-4f89-a60b-4278a0f3c562","year":2025},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"cited_paper":"/paper/2508.02324","citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:cffca4d4a4182208c9daad5435933dd6221efe8425e1ee055c9b3ac374b08be9","observation_id":"39cc047d-d3a3-460f-9086-6a79716f2359","resolution":{"observed_at":"2026-05-10T14:29:07.137498Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-05-21T15:23:05.016381+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-21T15:23:05.016381+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.18871","last_updated":"2026-04-21T17:32:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-23T17:38:54Z","title":"OmniGen2: Towards Instruction-Aligned Multimodal Generation","version":4},"cited_work":{"arxiv_id":"2506.18871","doi":null,"metadata_source":"pith","pith_arxiv_id":"2506.18871","snapshot_observed_at":"2026-07-10T01:46:41.215139Z","title":"OmniGen2: Towards Instruction-Aligned Multimodal Generation","venue":"cs.CV","work_id":"d3153e5f-b6e2-4ab3-9f41-e24e24d64496","year":2025},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"cited_paper":"/paper/2506.18871","citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:4ed8dc398ac6306bc6ff90985eb83a05c6199bcf61fa7ee9db3922f54ae52e48","observation_id":"4bb6c475-8721-4cde-92b0-1ecc64bc29cf","resolution":{"observed_at":"2026-05-10T00:49:48.824893Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.23747","last_updated":"2026-05-19T02:23:16Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-29T17:59:04Z","title":"Spatial-MLLM: Boosting MLLM Capabilities in Visual-based Spatial Intelligence","version":2},"cited_work":{"arxiv_id":"2505.23747","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.23747","snapshot_observed_at":"2026-07-04T16:39:57.786059Z","title":"Spatial-MLLM: Boosting MLLM Capabilities in Visual-based Spatial Intelligence","venue":"cs.CV","work_id":"389bb6a7-1369-4d6f-a808-838fa4f5b635","year":2025},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"cited_paper":"/paper/2505.23747","citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:13329276c3aa9e9a1d36327197a2988dacbfd2dea024e9de214e77e21a59df67","observation_id":"99c2eead-e394-43aa-9d7e-5d1a31091c07","resolution":{"observed_at":"2026-05-16T08:34:37.082411Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12528","last_updated":"2025-09-08T02:42:57Z","snapshot_observed_at":"2026-07-06T19:04:43.716629Z","submitted_at":"2024-08-22T16:32:32Z","title":"Show-o: One Single Transformer to Unify Multimodal Understanding and Generation","version":7},"cited_work":{"arxiv_id":"2408.12528","doi":"10.48550/arxiv.2408.12528","metadata_source":"pith","pith_arxiv_id":"2408.12528","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Show-o: One Single Transformer to Unify Multimodal Understanding and Generation","venue":"cs.CV","work_id":"1393dc24-a6b2-44e1-b5d7-7009d1fa4811","year":2024},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"cited_paper":"/paper/2408.12528","citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:5cf02c4d9bc4d14365edf508011bb5370a6b46edab4ceb39c9d43921d1991793","observation_id":"918548ed-3b9c-4015-aca2-0944ce8e6000","resolution":{"observed_at":"2026-05-11T21:03:33.779621Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.15564","last_updated":"2025-09-22T01:24:29Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-06-18T15:39:15Z","title":"Show-o2: Improved Native Unified Multimodal Models","version":3},"cited_work":{"arxiv_id":"2506.15564","doi":"10.48550/arxiv.2506.15564","metadata_source":"pith","pith_arxiv_id":"2506.15564","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Show-o2: Improved Native Unified Multimodal Models","venue":"cs.CV","work_id":"77f00563-1ce6-4fba-9d4e-c8ce83f716ac","year":2025},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"cited_paper":"/paper/2506.15564","citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:5b0a2f47360f8ad5446b660db6bc7ad64b730e2d4d077869fa00edbaa2d7b1db","observation_id":"f4cb575c-7395-4e90-8338-95c47cbe0b21","resolution":{"observed_at":"2026-05-12T18:51:16.424084Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"cited_work":{"arxiv_id":"2505.09388","doi":"10.1016/j.aiopen.2022.12","metadata_source":"pith","pith_arxiv_id":"2505.09388","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Qwen3 Technical Report","venue":"cs.CL","work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e","year":2025},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"cited_paper":"/paper/2505.09388","citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:56aeca59ad10b4b938d8173cdb27111330bc939b566d598fc8c1f8444a80c15d","observation_id":"6a29407a-c8c4-4f00-b289-68b6db70d18b","resolution":{"observed_at":"2026-05-10T00:49:48.840953Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Thinking in space: How mul- timodal large language models see, remember, and recall spaces","venue":null,"work_id":"a787f5a1-7729-4da6-b3e2-a4eaeaac7ce9","year":2025},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:58b986d89288fbd006336d53adc691b783be99324e292bcce3ea1f85ed61ea0f","observation_id":"46692265-b1b4-44e4-9d06-06ed7ab34d8d","resolution":{"observed_at":"2026-05-23T10:42:51.986046Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Learning interactive real-world simulators","venue":null,"work_id":"e52f715f-bef8-42f9-85d0-62c14a3ef4fd","year":2024},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:c9b59776e5d67c13277d6a746bb4d3bc80fd8ea5284a4cc6e8b1b5c2aa5b8206","observation_id":"2a70fb15-1b1b-4191-9a30-f71584fe65b5","resolution":{"observed_at":"2026-05-23T10:42:51.952957Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Scannet++: A high-fidelity dataset of 3d in- door scenes","venue":null,"work_id":"bc3c2baf-4dec-4a09-9a66-1cca6bede13d","year":2023},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:ecb6e190cdd3349836a68b5956289de352b6bec612f3bec9f11b4253ea26921c","observation_id":"08830a14-ec50-4e77-b707-bb259f4e5aaa","resolution":{"observed_at":"2026-05-23T10:42:52.016163Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2506.21458","doi":"10.48550/arxiv.2506.21458","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Spa- tial mental modeling from limited views","venue":"arXiv (Cornell University)","work_id":"e7da71d2-40fb-4670-9c50-b37e7469d519","year":2025},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:90b6e308c9d14b62b649ad6bf6760cb319ef68fc7a51665374d4d19e707eda6c","observation_id":"dfc0bdae-e1bc-4e93-9bad-5c1ce9021bbd","resolution":{"observed_at":"2026-05-10T00:49:48.869102Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.10721","last_updated":"2024-06-15T19:22:51Z","snapshot_observed_at":"2026-07-06T18:31:36.504766Z","submitted_at":"2024-06-15T19:22:51Z","title":"RoboPoint: A Vision-Language Model for Spatial Affordance Prediction for Robotics","version":1},"cited_work":{"arxiv_id":"2406.10721","doi":"10.48550/arxiv.2406.10721","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.10721","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Robopoint: A vision-language model for spatial affordance prediction for robotics","venue":"arXiv (Cornell University)","work_id":"af457b2b-ad31-4178-ae3c-b18aaa2623ec","year":2024},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"cited_paper":"/paper/2406.10721","citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:49c8c24f20a78c3c17f51396785d85326c1b87b473ba254dbe734ceb0b0fa9c1","observation_id":"c3db3868-3eeb-4fb3-845e-da793759f331","resolution":{"observed_at":"2026-05-10T00:49:48.855770Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2504.07958","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T16:29:57.552320Z","title":"Zhang, H","venue":null,"work_id":"d3a1ce24-6135-4652-827f-596adf7c2220","year":2025},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:3a079e4e21ef54635f20d1ca8bbe07c48017b680016b2b61c1c00fb97d843c6c","observation_id":"ed3d13da-dde6-40d3-a300-60759e9ac045","resolution":{"observed_at":"2026-05-10T00:49:48.838326Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2509.12129","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-11T02:27:49.210222Z","title":"Embodied navigation foundation model","venue":null,"work_id":"12f41b76-9aed-4727-a532-07029f19a43c","year":2025},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:62a623bcffef46008955cd9e4f7130568a63f5368c830c12ce8c12e7ba4f46dd","observation_id":"da0e6675-788e-4866-9996-a791a6300cab","resolution":{"observed_at":"2026-05-10T00:49:48.843653Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T03:15:58.030192Z","title":"The unreasonable effectiveness of deep features as a perceptual metric","venue":null,"work_id":"a1b1bbd9-f486-4890-97fd-9338b7f32312","year":2018},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:bce7dc35c714dee18a2aa63e94c0066f00a487756cf844ebb4d1203d6aecc542","observation_id":"68db66cb-4be5-49f6-acd5-a6ebeea49058","resolution":{"observed_at":"2026-05-23T10:42:51.966115Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Ultraedit: Instruction-based fine-grained image editing at scale.Advances in Neural Information Pro- cessing Systems, 37:3058–3093","venue":null,"work_id":"d12149ca-eabe-4fde-8fcb-61196056f228","year":2024},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:d909b9d937cd3a55791a29051cbf64e2f60fc143fd5d25bd8533acfaba37007a","observation_id":"05532073-2050-433e-b8b9-5f65574678ad","resolution":{"observed_at":"2026-05-23T10:42:51.972212Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Towards learning a generalist model for embod- ied navigation","venue":null,"work_id":"8313f262-f31a-499d-8844-fa2b81022a91","year":2024},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:fdf7c5ba1ab6656ca87314625405d713d8f6ca6d0bc5f268ff2761cd5b42fd10","observation_id":"8a4422f8-688b-49a9-8ae3-43b4ac402bf8","resolution":{"observed_at":"2026-05-23T10:42:52.005264Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.10813","last_updated":"2026-04-28T04:11:01Z","snapshot_observed_at":"2026-07-06T22:29:38.297334Z","submitted_at":"2025-09-13T14:25:17Z","title":"InternScenes: A Large-scale Simulatable Indoor Scene Dataset with Realistic Layouts","version":4},"cited_work":{"arxiv_id":"2509.10813","doi":null,"metadata_source":"pith","pith_arxiv_id":"2509.10813","snapshot_observed_at":"2026-07-10T23:17:45.647946Z","title":"InternScenes: A Large-scale Simulatable Indoor Scene Dataset with Realistic Layouts","venue":"cs.CV","work_id":"d60a2268-6de6-4160-b967-d34c5e76b7f3","year":2025},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"cited_paper":"/paper/2509.10813","citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:1efcb5e87c5f34aa6e209d735eab125161b41a80e14a681bdde99e53e7d04e47","observation_id":"9d8464aa-abef-4063-91bd-259a084fb386","resolution":{"observed_at":"2026-05-10T00:49:48.827500Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Robotrom-nav: A unified frame- work for embodied navigation integrating perception, plan- ning, and prediction","venue":null,"work_id":"07fda2c1-5bc2-4a14-9ee0-8a19d1866eb9","year":2025},"citing_paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-10T00:45:46.261005Z"},"links":{"citing_paper":"/paper/2604.20570"},"observation_digest":"sha256:5640747aed504b5a2655aa4912750efcec78b28177231e624452f81372b452d9","observation_id":"167c9bac-4c21-457a-8ca8-bfafbde8ce8e","resolution":{"observed_at":"2026-05-23T10:42:51.946439Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2604.20570","last_updated":"2026-04-22T13:50:00Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-07-06T23:07:03.254122Z","submitted_at":"2026-04-22T13:50:00Z","title":"Exploring Spatial Intelligence from a Generative Perspective"},"reference_resolution":{"displayed":49,"state_counts":{"malformed_identifier":0,"metadata_mismatch":3,"parse_uncertain":0,"unresolved":0,"verified_exact":32,"verified_fuzzy":14},"total_outbound_references":49},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 49 of 49 outbound references and 0 inbound Pith citation observations for arXiv:2604.20570."}