{"as_of":"2026-08-07T23:10:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:48d692f2a827712c399cd510dcb9507ccb9444eaf3ddd68cdfb67b016af2ed8b","coverage":[{"denominator":66,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":66,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T21:52:25.446679Z","state":"measured"},{"denominator":68,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":68,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-28T06:58:39.117228Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-02T07:26:45.809288Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"cited_work":{"arxiv_id":"2506.23219","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.23219","snapshot_observed_at":"2026-07-02T07:26:45.809288Z","title":"https://api.semanticscholar.org/CorpusID: 280010693","venue":null,"work_id":"99e1ebd3-1e7b-4a41-a793-35ef7ac42029","year":2025},"citing_paper":{"arxiv_id":"2604.08033","last_updated":"2026-04-09T09:38:15Z","snapshot_observed_at":"2026-07-06T22:57:13.745326Z","submitted_at":"2026-04-09T09:38:15Z","title":"IoT-Brain: Grounding LLMs for Semantic-Spatial Sensor Scheduling","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-10T17:49:16.409834Z"},"links":{"cited_paper":"/paper/2506.23219","citing_paper":"/paper/2604.08033"},"observation_digest":"sha256:a3ef020647a09e49f3bc4c7b6a5ced157b6166ec14ae0da7336e9f5a64f7e4fd","observation_id":"eba8392f-a4ca-4c0c-b565-bdf2db73376e","resolution":{"observed_at":"2026-05-11T06:05:56.875660Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"cited_work":{"arxiv_id":"2506.23219","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.23219","snapshot_observed_at":"2026-07-02T07:26:45.809288Z","title":"https://api.semanticscholar.org/CorpusID: 280010693","venue":null,"work_id":"99e1ebd3-1e7b-4a41-a793-35ef7ac42029","year":2025},"citing_paper":{"arxiv_id":"2606.04381","last_updated":"2026-06-03T02:54:59Z","snapshot_observed_at":"2026-07-06T23:44:28.265478Z","submitted_at":"2026-06-03T02:54:59Z","title":"From Symbolic to Geometric: Enabling Spatial Reasoning in Large Language Models","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-28T06:58:39.117228Z"},"links":{"cited_paper":"/paper/2506.23219","citing_paper":"/paper/2606.04381"},"observation_digest":"sha256:7621f2c77a7df7887a4ed70114e650c7e9fca797e5311eb7082e8cbc3b1b7a8b","observation_id":"4a8bd704-34ac-4f67-ab3b-97c1b8c68f6c","resolution":{"observed_at":"2026-07-02T07:26:45.810959Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2506.23219/citation-record","integrity":"/paper/2506.23219/integrity","json":"/paper/2506.23219/citation-record.json","paper":"/paper/2506.23219"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2403.09059","last_updated":"2024-11-12T06:15:50Z","snapshot_observed_at":"2026-08-07T21:43:15.197626Z","submitted_at":"2024-03-14T02:56:38Z","title":"LAMP: A Language Model on the Map","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.09059","snapshot_observed_at":"2026-08-06T21:52:19.753815Z","title":"Lamp: A language model on the map","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:19.753815Z"},"links":{"cited_paper":"/paper/2403.09059","citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:e9e21c75b7b12cfb7f1477d6ba6f2a30bd8afb79f6d710b143447abd98b5f955","observation_id":"a66870f0-4af1-4c6e-b323-672ffb45e0fd","resolution":{"observed_at":"2026-08-06T21:52:19.753815Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:34.808717Z","title":"City foundation models for learning general purpose rep- resentations from openstreetmap","venue":null,"work_id":"d0cff2ae-7aba-457d-a277-7f67f1acbd42","year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:19.839179Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:2b2657a4a0eb7461cda43ec1f69fc5ec85f3e1560eaf01e898c5466c82146b69","observation_id":"69493768-4882-459e-b673-245ff3654219","resolution":{"observed_at":"2026-08-06T21:52:34.883775Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:34.620458Z","title":"Street view imagery in urban analytics and gis: A review","venue":null,"work_id":"cac9e22e-a157-4c47-8bfb-f1d3da2eba56","year":2021},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:19.924010Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:8176536507daf33bff73be96eb59d14aff55e207aef2adc8c805a48c5feb0206","observation_id":"983a5a70-1d8b-48e3-99a8-3f7c8161bcbc","resolution":{"observed_at":"2026-08-06T21:52:34.689303Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15818","last_updated":"2023-07-28T21:18:02Z","snapshot_observed_at":"2026-08-02T16:17:50.621617Z","submitted_at":"2023-07-28T21:18:02Z","title":"RT-2: Vision-Language-Action Models Transfer Web Knowledge to Robotic Control","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15818","snapshot_observed_at":"2026-08-06T21:52:20.060217Z","title":"Rt-2: Vision-language-action models transfer web knowledge to robotic control","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:20.060217Z"},"links":{"cited_paper":"/paper/2307.15818","citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:4dba658a12bc9ea59266f4970b2b3c3c277df5c8aa44107fe232dc1a4d71b455","observation_id":"10e1defb-e8da-4ac3-88bc-9edc797d602c","resolution":{"observed_at":"2026-08-06T21:52:20.060217Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:34.427094Z","title":"Touchdown: Natural language naviga- tion and spatial reasoning in visual street environments","venue":null,"work_id":"93ed5db4-3b79-4cf6-b4f7-ae90cf4e2c00","year":2019},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:20.178492Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:073e44452882c044f80c6f8bd5e47a9274115aac4d191e0e8deb8123d492a231","observation_id":"bfce9a89-df1e-437e-b8b9-a9c1ed8543ae","resolution":{"observed_at":"2026-08-06T21:52:34.487883Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.12793","last_updated":"2023-11-28T08:52:50Z","snapshot_observed_at":"2026-08-04T08:17:54.774738Z","submitted_at":"2023-11-21T18:58:11Z","title":"ShareGPT4V: Improving Large Multi-Modal Models with Better Captions","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.12793","snapshot_observed_at":"2026-08-06T21:52:20.279386Z","title":"Sharegpt4v: Improving large multi-modal models with better captions","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:20.279386Z"},"links":{"cited_paper":"/paper/2311.12793","citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:53a4bd22b93020599db251a7de146d0bc9e10cf4e0d2400943e891d9976908e7","observation_id":"c0282037-4302-4a00-acf3-f66c822daa7d","resolution":{"observed_at":"2026-08-06T21:52:20.279386Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.16821","last_updated":"2024-04-29T20:24:30Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-04-25T17:59:19Z","title":"How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.16821","snapshot_observed_at":"2026-08-06T21:52:20.373110Z","title":"How far are we to gpt-4v? closing the gap to commercial multimodal models with open-source suites","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:20.373110Z"},"links":{"cited_paper":"/paper/2404.16821","citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:fe94725bbe9f9254901dddda1650102e1efe0951c3e1c5a4123b88ca03a058ca","observation_id":"f5b2046a-852f-4c85-b199-852e4af903d3","resolution":{"observed_at":"2026-08-06T21:52:20.373110Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:34.267818Z","title":"Internvl: Scaling up vision founda- tion models and aligning for generic visual-linguistic tasks","venue":null,"work_id":"576176a2-1fec-4d0e-878f-acb8121dfea7","year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:20.462260Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:5d0c19080f2ec1f96dec3a4dae1d1015756d3d3df326d9b82a34574255b740d7","observation_id":"44fc0a9b-b231-4285-8142-b90368f7e855","resolution":{"observed_at":"2026-08-06T21:52:34.338270Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01584","last_updated":"2024-10-15T01:16:20Z","snapshot_observed_at":"2026-07-06T18:24:38.958815Z","submitted_at":"2024-06-03T17:59:06Z","title":"SpatialRGPT: Grounded Spatial Reasoning in Vision Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01584","snapshot_observed_at":"2026-08-06T21:52:20.579394Z","title":"Spatial- rgpt: Grounded spatial reasoning in vision language model","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:20.579394Z"},"links":{"cited_paper":"/paper/2406.01584","citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:14566d4e210fbdd4661ebd137d45b588a36e88984f0fe3c684334bd091d6a74c","observation_id":"ca1bd479-40fe-484f-a892-1beea35e5999","resolution":{"observed_at":"2026-08-06T21:52:20.579394Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:20.691552Z","title":"Understanding world or predict- ing future? a comprehensive survey of world models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:20.691552Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:c43c1a66d875d04d62252962cff56eceffeb714e943767140a858bfe0006bea4","observation_id":"d5635dfc-a5bd-429a-a4ca-67bd4b671ad8","resolution":{"observed_at":"2026-08-06T21:52:20.691552Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.14233","last_updated":"2023-05-23T16:49:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-23T16:49:14Z","title":"Enhancing Chat Language Models by Scaling High-quality Instructional Conversations","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.14233","snapshot_observed_at":"2026-08-06T21:52:20.776159Z","title":"Enhancing chat language models by scal- ing high-quality instructional conversations","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:20.776159Z"},"links":{"cited_paper":"/paper/2305.14233","citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:162512f92d80118edbd4c2e63709b0e33521b2d54506aac371eaa3b3e6b5907f","observation_id":"f8f69a58-ccc7-48f3-94c7-4d997701a538","resolution":{"observed_at":"2026-08-06T21:52:20.776159Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.05492","last_updated":"2024-06-07T15:51:08Z","snapshot_observed_at":"2026-07-06T16:29:39.982733Z","submitted_at":"2023-10-09T07:56:16Z","title":"How Abilities in Large Language Models are Affected by Supervised Fine-tuning Data Composition","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.05492","snapshot_observed_at":"2026-08-06T21:52:20.860831Z","title":"How abilities in large lan- guage models are affected by supervised fine-tuning data composition","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:20.860831Z"},"links":{"cited_paper":"/paper/2310.05492","citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:ba8d981679555457577ec697ebbf73a7b495dfd486e75e00bdc2bdc8f58f6b6b","observation_id":"b09e2d68-1545-4e28-b366-9000934f1574","resolution":{"observed_at":"2026-08-06T21:52:20.860831Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:20.969557Z","title":"Vlmevalkit: An open- source toolkit for evaluating large multi-modality models,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:20.969557Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:b354a7ac2b9be8a00e7c43b4cba8bb476fdf00c941d622c8c9a196d91cc09170","observation_id":"33b27999-3ad1-4f5a-ad15-fbb09880377a","resolution":{"observed_at":"2026-08-06T21:52:20.969557Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:34.093097Z","title":"Urban visual intelligence: Uncovering hidden city pro- files with street view images","venue":null,"work_id":"6ac9549e-e54e-4a39-92e5-2718f65f7ba9","year":2023},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:21.035326Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:59de91cb3dc6c86c2e2ff570e27503b2e714026dc2d2d0c7c47fe0989d3de2fa","observation_id":"1b813eca-e9b5-4467-99db-6e324c9003b4","resolution":{"observed_at":"2026-08-06T21:52:34.158394Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:33.948118Z","title":"Agent- move: A large language model based agentic framework for zero-shot next location prediction","venue":null,"work_id":"fec6fe57-5d3f-43f6-b9b5-affb50ca1e37","year":2025},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:21.152105Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:785ed75da83f7d5705336ba3dac83d6df25b3ed76bf673d4b7207ec88b27b0b5","observation_id":"d719cf63-a281-4dc5-b9a6-fbf82669f95c","resolution":{"observed_at":"2026-08-06T21:52:34.016597Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:33.693058Z","title":"Citygpt: Empowering urban spatial cognition of large language models","venue":null,"work_id":"2706224d-90c5-458e-aebc-6c744045843a","year":2025},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:21.237329Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:0d6215dfb70074711889d886615fbb56bdaec01505ee1e7b8c49449f61c62cc5","observation_id":"5ab581ac-0389-4ed1-a2ef-94ca7df1bca1","resolution":{"observed_at":"2026-08-06T21:52:33.818015Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.09848","last_updated":"2025-04-14T03:38:31Z","snapshot_observed_at":"2026-08-07T16:06:02.866643Z","submitted_at":"2025-04-14T03:38:31Z","title":"A Survey of Large Language Model-Powered Spatial Intelligence Across Scales: Advances in Embodied Agents, Smart Cities, and Earth Science","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.09848","snapshot_observed_at":"2026-08-06T21:52:21.335105Z","title":"A survey of large language model-powered spatial intelligence across scales: Advances in embodied agents, smart cities, and earth science","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:21.335105Z"},"links":{"cited_paper":"/paper/2504.09848","citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:09b9afad18f7a39e7b75af48fd912e38d87af9b7d2257770ba61fbcdda363587","observation_id":"a030dcfb-3b62-4f79-8cd5-b941de0a9c60","resolution":{"observed_at":"2026-08-06T21:52:21.335105Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:33.500695Z","title":"City- bench: Evaluating the capabilities of large language models for urban tasks","venue":null,"work_id":"32e0dbbb-c9e0-4bb0-84a5-f1f6243d570f","year":2025},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:21.403131Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:7e7ae638c0fd4d2ce8afea142cd7d14c8a4bf58d87a23043110d005fb53a39e5","observation_id":"50e01ba8-4998-46e5-b4d7-22fb6e6410dc","resolution":{"observed_at":"2026-08-06T21:52:33.556275Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:21.502930Z","title":"Imagebind: One embedding space to bind them all","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:21.502930Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:b13587994df5b5840f11e7007828c749d232efc6d947184dc9851ea8ac4e2230","observation_id":"dc817789-0086-487f-a05f-52dacc81f7ab","resolution":{"observed_at":"2026-08-06T21:52:21.502930Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.00823","last_updated":"2024-10-29T01:58:06Z","snapshot_observed_at":"2026-07-06T19:43:46.557944Z","submitted_at":"2024-10-29T01:58:06Z","title":"Mobility-LLM: Learning Visiting Intentions and Travel Preferences from Human Mobility Data with Large Language Models","version":1},"cited_work":{"arxiv_id":"2411.00823","doi":null,"metadata_source":"pith","pith_arxiv_id":"2411.00823","snapshot_observed_at":"2026-08-06T21:52:25.934068Z","title":"Mobility-LLM: Learning Visiting Intentions and Travel Preferences from Human Mobility Data with Large Language Models","venue":"cs.LG","work_id":"40f599c5-4b46-4b81-84c1-dc7f3ca8e6ef","year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:21.581742Z"},"links":{"cited_paper":"/paper/2411.00823","citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:fabb85b693e391b7538f11199e073516c9650feee68dea5444f7f6443786ba8b","observation_id":"4f1933ac-cb85-483b-b11f-7834f2a98765","resolution":{"observed_at":"2026-08-06T21:52:26.087467Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:33.295368Z","title":"Regiongpt: Towards region understanding vision lan- guage model","venue":null,"work_id":"0edebe3b-1c7f-4975-9d4e-e0b44fcd0eba","year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:21.665075Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:4441b73322fca9d7075e248f68d9ec1bf95a30937a7a4e504089cf3a3c57c9a6","observation_id":"f61311cd-f3e4-4df6-b205-1b23b4838077","resolution":{"observed_at":"2026-08-06T21:52:33.357391Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.16831","last_updated":"2025-01-22T08:45:56Z","snapshot_observed_at":"2026-07-06T17:50:13.914184Z","submitted_at":"2024-03-25T14:57:18Z","title":"UrbanVLP: Multi-Granularity Vision-Language Pretraining for Urban Socioeconomic Indicator Prediction","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.16831","snapshot_observed_at":"2026-08-06T21:52:21.758851Z","title":"Urbanvlp: A multi- granularity vision-language pre-trained foundation model for urban indicator prediction","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:21.758851Z"},"links":{"cited_paper":"/paper/2403.16831","citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:25b2ba17e9fd47b95227b78f231dd6cafeb9887517203833b487f165dea3354e","observation_id":"57e99a67-4c0a-4127-8b1c-d71f82485cd5","resolution":{"observed_at":"2026-08-06T21:52:21.758851Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:33.084582Z","title":"Vision-language models for medical report generation and visual question answering: A review, 2024","venue":null,"work_id":"b70cb0f6-9855-4790-9124-24b6dc204258","year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:21.863618Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:f8ed5ba9e0499e63ff00f27526976321a50049b7fb3742f5abb348fa9406d578","observation_id":"8e59aa17-e176-46fe-b9ea-ac00f6d93b75","resolution":{"observed_at":"2026-08-06T21:52:33.222258Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.15266","last_updated":"2023-07-28T02:23:35Z","snapshot_observed_at":"2026-08-05T09:46:05.780158Z","submitted_at":"2023-07-28T02:23:35Z","title":"RSGPT: A Remote Sensing Vision Language Model and Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.15266","snapshot_observed_at":"2026-08-06T21:52:21.928081Z","title":"Rsgpt: A remote sensing vision language model and benchmark","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:21.928081Z"},"links":{"cited_paper":"/paper/2307.15266","citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:253ec3658b8a1a6bd62866a26f584c9b0d521cc2e95cbbefb2f2ecc2d148d5cf","observation_id":"89386da8-4e18-4d7c-801e-a4d5714a0262","resolution":{"observed_at":"2026-08-06T21:52:21.928081Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01728","last_updated":"2024-01-29T06:27:53Z","snapshot_observed_at":"2026-08-04T04:31:27.172482Z","submitted_at":"2023-10-03T01:31:25Z","title":"Time-LLM: Time Series Forecasting by Reprogramming Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01728","snapshot_observed_at":"2026-08-06T21:52:22.043641Z","title":"Time-llm: Time series forecasting by reprogramming large language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:22.043641Z"},"links":{"cited_paper":"/paper/2310.01728","citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:0d0e9d88d3648bfcc267858896df44570876ec830ef73d24a2fe7b7f0ebc4064","observation_id":"4c330697-6e82-442f-9a38-205e3a50ed8c","resolution":{"observed_at":"2026-08-06T21:52:22.043641Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:32.910871Z","title":"Geochat: Grounded large vision-language model for remote sensing","venue":null,"work_id":"885dbdc0-ecf6-458e-849d-ab12815ed80c","year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:22.119635Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:0bb8bf3aa86a5c3c0deb3e8479b221a04092857412e871ee5c85c1b59f31060a","observation_id":"969e79a5-5099-4815-b454-9f78d05a7ada","resolution":{"observed_at":"2026-08-06T21:52:32.998699Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:32.723289Z","title":"Llava-med: Training a large language- and-vision assistant for biomedicine in one day.Advances in Neural Information Processing Systems, 36, 2024","venue":null,"work_id":"234ff8e3-f1f5-45e0-9f3b-6c56d84498ed","year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:22.243892Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:536a7433b69531c8c8099d59ed1bcdc377f3e5baf981248aff2f5c8295e1a786","observation_id":"1899628e-54f6-4046-8779-ac2b441d92f0","resolution":{"observed_at":"2026-08-06T21:52:32.803628Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:32.539200Z","title":"Urbangpt: Spatio- temporal large language models","venue":null,"work_id":"660b244a-3cbe-40ca-878e-2ebe09e5f0e7","year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:22.380917Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:4163333e21cfa4110c0985fd0627cb4632402bd9b140d507fd616c33e8817dfc","observation_id":"0a1756f7-099d-4bc0-81f1-af3980b310af","resolution":{"observed_at":"2026-08-06T21:52:32.628393Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:32.366052Z","title":"Vila: On pre-training for visual language models","venue":null,"work_id":"a27edd05-2a5d-40ba-bc9d-5bd99e9dfd28","year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:22.486258Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:142b43f09d5ea21dfa78099f3b32c846677f6c72f7357a26c7eb14fc19f20099","observation_id":"7fcd0e7e-db48-4ec4-8f6f-70a57a09de3e","resolution":{"observed_at":"2026-08-06T21:52:32.420034Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:32.185801Z","title":"Improved baselines with visual instruction tuning","venue":null,"work_id":"910fbf88-58ad-48dc-b5e8-35fddb27d548","year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:22.603566Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:e017ee3d94885ab96fbc4576dfdb067768539c8f8acb122b0dfe206dbd87d329","observation_id":"e414ca83-90fd-4e0b-a0ae-fe4fa6fc5785","resolution":{"observed_at":"2026-08-06T21:52:32.256034Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:32.005820Z","title":"Visual instruction tuning","venue":null,"work_id":"6cba9273-53f4-4278-8dff-af868cce82f5","year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:22.687759Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:0211126520cbbef4489e8c9532ad7c786d7c877405d03170650a3ce8d7e6e5ff","observation_id":"774b3427-2c81-4631-9e66-edd2016f0633","resolution":{"observed_at":"2026-08-06T21:52:32.082401Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:22.798308Z","title":"Citylens: Bench- 10 marking large language-vision models for urban socioeco- nomic sensing","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:22.798308Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:dc00b5465f0d727325f1f3b7a835a16f51017e55d1a022bbe4c3dcafbad5ebad","observation_id":"e39f3b66-cf9c-4426-963e-f6fafe0362c4","resolution":{"observed_at":"2026-08-06T21:52:22.798308Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.10100","last_updated":"2024-07-08T04:33:37Z","snapshot_observed_at":"2026-07-06T18:31:04.425917Z","submitted_at":"2024-06-14T14:57:07Z","title":"SkySenseGPT: A Fine-Grained Instruction Tuning Dataset and Model for Remote Sensing Vision-Language Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.10100","snapshot_observed_at":"2026-08-06T21:52:22.909176Z","title":"Skysensegpt: A fine-grained in- struction tuning dataset and model for remote sensing vision- language understanding","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:22.909176Z"},"links":{"cited_paper":"/paper/2406.10100","citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:de86888b93423f532ba8d980d93ae1ccff5167711a2c25ebb71bc956742425e7","observation_id":"cf9c645e-b765-4ad7-85d1-1def9183b5d5","resolution":{"observed_at":"2026-08-06T21:52:22.909176Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.00438","last_updated":"2023-12-01T09:10:33Z","snapshot_observed_at":"2026-07-06T16:55:33.394778Z","submitted_at":"2023-12-01T09:10:33Z","title":"Dolphins: Multimodal Language Model for Driving","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.00438","snapshot_observed_at":"2026-08-06T21:52:23.026538Z","title":"Dolphins: Multimodal language model for driving","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:23.026538Z"},"links":{"cited_paper":"/paper/2312.00438","citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:dd0ba7a6437fc3b7c38887f3753038856385dd0bdd3887fb01656767c64052f0","observation_id":"6dd443bf-64e6-47c3-be1f-8389e6fb831e","resolution":{"observed_at":"2026-08-06T21:52:23.026538Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:31.834802Z","title":"On the opportunities and chal- lenges of foundation models for geoai (vision paper)","venue":null,"work_id":"91cf7ae0-6b4f-4676-a997-f966b6937573","year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:23.139438Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:a13a39417e5dddb440aba3f1ebf44e11dc9530e13a5c35557dfa9bfd7198f666","observation_id":"335ccf75-0269-4d48-831b-104bd6031418","resolution":{"observed_at":"2026-08-06T21:52:31.920846Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:31.631802Z","title":"LLaMA 3.2: Advancing Vision, Edge, and Mo- bile Devices","venue":null,"work_id":"0218bf09-5753-406c-a3f1-e80412093b59","year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:23.201817Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:2031be2c370d5da01964ede79e3297e5481ec33227635b998db6c9e7ad790466","observation_id":"188d55e0-3b17-4ea4-bc96-86149bdc97ce","resolution":{"observed_at":"2026-08-06T21:52:31.727923Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.02544","last_updated":"2024-07-16T01:40:34Z","snapshot_observed_at":"2026-08-05T08:18:10.815365Z","submitted_at":"2024-02-04T15:46:43Z","title":"LHRS-Bot: Empowering Remote Sensing with VGI-Enhanced Large Multimodal Language Model","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.02544","snapshot_observed_at":"2026-08-06T21:52:23.293073Z","title":"Lhrs-bot: Empowering remote sensing with vgi-enhanced large multimodal language model","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:23.293073Z"},"links":{"cited_paper":"/paper/2402.02544","citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:a9ce0d0f0a34c74a1a9a987099f0f2d7a4f9d0dbcfa81c362f01958735629567","observation_id":"2bc776dd-464a-45e3-b098-c2300c8c73d3","resolution":{"observed_at":"2026-08-06T21:52:23.293073Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:31.423724Z","title":"Introducing chatgpt","venue":null,"work_id":"5fb89827-bb14-40a0-b112-c1ce56b911c1","year":2022},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:23.383898Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:ffef1b00065c467770b4826ce88baf90aeb34bc1db1a9aec8f5a579e5c7acd66","observation_id":"bccbedbe-3ea0-47a8-a30b-31564a8b6581","resolution":{"observed_at":"2026-08-06T21:52:31.520569Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:31.220882Z","title":"Gpt-4v(ision) system card","venue":null,"work_id":"e3f97a34-3536-4159-b6ed-781d4bc828cb","year":2023},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:23.464642Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:0813d06a3021eaaed73e1643731ab59fd9f1dbc2904f514ea4b9b13e03eec5d3","observation_id":"8013e578-b61f-4c03-8585-db60f194e6e8","resolution":{"observed_at":"2026-08-06T21:52:31.306992Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:31.035922Z","title":"Hello GPT-4","venue":null,"work_id":"0a607a22-3132-4584-8897-839e5c5f92e6","year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:23.579833Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:59872267cd44ffcf5cc91ec47c86431f5887ab198888ca483fd2887a934648c7","observation_id":"35a0cb28-947f-4e9c-842a-a0b494a6e2ea","resolution":{"observed_at":"2026-08-06T21:52:31.119217Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12191","last_updated":"2024-10-03T15:54:49Z","snapshot_observed_at":"2026-08-06T05:35:29.109022Z","submitted_at":"2024-09-18T17:59:32Z","title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12191","snapshot_observed_at":"2026-08-06T21:52:23.674340Z","title":"Qwen2-vl: Enhancing vision-language model’s perception of the world at any resolution","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:23.674340Z"},"links":{"cited_paper":"/paper/2409.12191","citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:a7509637bfa7377e629b86a88abf9e3d6ae6f7bbed888092d2392f0c09d31787","observation_id":"985fb899-f005-463c-b1e5-e3499e64c72f","resolution":{"observed_at":"2026-08-06T21:52:23.674340Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.08394","last_updated":"2024-12-31T05:35:05Z","snapshot_observed_at":"2026-08-06T17:31:18.801416Z","submitted_at":"2024-06-12T16:44:50Z","title":"VisionLLM v2: An End-to-End Generalist Multimodal Large Language Model for Hundreds of Vision-Language Tasks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.08394","snapshot_observed_at":"2026-08-06T21:52:23.743835Z","title":"Visionllm v2: An end-to-end general- ist multimodal large language model for hundreds of vision- language tasks","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:23.743835Z"},"links":{"cited_paper":"/paper/2406.08394","citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:14bdb29bce716635435b898ae95e67dc885198874082a6991106175a7cd3a129","observation_id":"bf746aaa-b2a3-428e-94c4-d7883a7ee9c8","resolution":{"observed_at":"2026-08-06T21:52:23.743835Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:30.873617Z","title":"V*: Guided visual search as a core mechanism in multimodal llms","venue":null,"work_id":"f0379d4a-5d63-4a48-9397-e8b1699ea968","year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:23.852374Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:5867136287a5b10383d5588189fbe919aa9e1573d3241a1156bb77166c9391ba","observation_id":"ed1fafd1-c4d0-420a-9d24-a1223f0adc7b","resolution":{"observed_at":"2026-08-06T21:52:30.950965Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:30.739046Z","title":"RealworldQA Dataset","venue":null,"work_id":"eb565141-2831-46ba-899b-0adb5a4cc1b7","year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:23.943108Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:59a27f9ca9756d704f4171d25ae0484c1299cd7b5a791853c18f0974c9e807d1","observation_id":"3f187676-0180-47b0-941b-b631ccba57d8","resolution":{"observed_at":"2026-08-06T21:52:30.803340Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:30.577454Z","title":"Analyz- ing large language models’ capability in location prediction","venue":null,"work_id":"591ebf3c-7735-49b9-bf7e-1f8e28158402","year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:24.066125Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:79d8ae3982297fe9c38b2fb6087833d1223243f3277c6102b28ac46cd94fec69","observation_id":"83b5e54f-4f66-4e79-a575-7d0fce68ee2b","resolution":{"observed_at":"2026-08-06T21:52:30.663944Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11813","last_updated":"2023-12-19T03:12:13Z","snapshot_observed_at":"2026-07-06T17:05:02.853583Z","submitted_at":"2023-12-19T03:12:13Z","title":"Urban Generative Intelligence (UGI): A Foundational Platform for Agents in Embodied City Environment","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11813","snapshot_observed_at":"2026-08-06T21:52:24.182301Z","title":"Ur- ban generative intelligence (ugi): A foundational platform for agents in embodied city environment","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:24.182301Z"},"links":{"cited_paper":"/paper/2312.11813","citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:0faebe670e0248794301c0613e6dfda4a8c832f36fb27d367fe39013bfba05d5","observation_id":"8dc2dfb8-1c7c-4968-8c4e-506a1669c93f","resolution":{"observed_at":"2026-08-06T21:52:24.182301Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.09686","last_updated":"2025-01-23T08:44:44Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-16T17:37:58Z","title":"Towards Large Reasoning Models: A Survey of Reinforced Reasoning with Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.09686","snapshot_observed_at":"2026-08-06T21:52:24.272058Z","title":"Towards large rea- soning models: A survey of reinforced reasoning with large language models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:24.272058Z"},"links":{"cited_paper":"/paper/2501.09686","citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:20fe470d87a38391a88ece4255d15116720c340a6cd8d08f875602aa71d5e00d","observation_id":"9df47446-7336-44b2-a638-89adaf931ff2","resolution":{"observed_at":"2026-08-06T21:52:24.272058Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:30.396042Z","title":"Par- ticipatory cultural mapping based on collective behavior data in location-based social networks","venue":null,"work_id":"3696a1ca-4930-4ae9-8acc-0a28cd0bf861","year":null},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:24.356229Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:51ac161a9110a960fb52fcd5d55f720d549ee29aff8a762b3627ae716284c736","observation_id":"a42cc69e-bb69-44fa-b228-7a09bb0316a3","resolution":{"observed_at":"2026-08-06T21:52:30.479881Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.13549","last_updated":"2024-11-29T15:51:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-23T15:21:52Z","title":"A Survey on Multimodal Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.13549","snapshot_observed_at":"2026-08-06T21:52:24.419530Z","title":"A survey on multimodal large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:24.419530Z"},"links":{"cited_paper":"/paper/2306.13549","citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:5e10b7f8867d97fc92ccf02b4ede4ff3e7566c6b5f5e371bd1efc86cb1116f72","observation_id":"af2619d3-6b9d-4e69-a7fa-59a751aa8b8f","resolution":{"observed_at":"2026-08-06T21:52:24.419530Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:30.217616Z","title":"Mm-vet: Evaluating large multimodal models for inte- grated capabilities","venue":null,"work_id":"cc8e35d6-0875-413c-8f2f-4c05f8f8c281","year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:24.484165Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:1d30dd7b85c25c1cfeb474041be5c994239ceff4bed212a8e3479c5cf2934804","observation_id":"cd0b7066-f87c-4c05-adf0-3d98eb6d0e1d","resolution":{"observed_at":"2026-08-06T21:52:30.296895Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.09712","last_updated":"2024-01-18T04:10:20Z","snapshot_observed_at":"2026-07-06T17:17:11.535092Z","submitted_at":"2024-01-18T04:10:20Z","title":"SkyEyeGPT: Unifying Remote Sensing Vision-Language Tasks via Instruction Tuning with Large Language Model","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.09712","snapshot_observed_at":"2026-08-06T21:52:24.557582Z","title":"Skyeyegpt: Unifying remote sensing vision-language tasks via instruc- tion tuning with large language model","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:24.557582Z"},"links":{"cited_paper":"/paper/2401.09712","citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:880430df2b08e14fbd716f8708cb3c222684f2e8764c7b5c16a31dac83722109","observation_id":"4ea88e8a-8db0-4ce1-8d73-77ef38fac776","resolution":{"observed_at":"2026-08-06T21:52:24.557582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:30.029521Z","title":"Earthgpt: A universal multi-modal large lan- guage model for multi-sensor image comprehension in re- mote sensing domain","venue":null,"work_id":"efce21b3-6278-4ee5-829c-6a869f142c18","year":2024},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:24.628535Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:3d5ecb9a0681aabc5677f2b6b4ef41f0d4305ca3e2e8b384d911aafe6d6eeadb","observation_id":"fb0d3055-79a3-4d39-af6d-94273c5476cc","resolution":{"observed_at":"2026-08-06T21:52:30.120055Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:29.897486Z","title":"Urban foundation models: A survey","venue":null,"work_id":"e698ec71-25b3-4844-a258-2e705472c7f0","year":null},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:24.672320Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:1e87173591771d898c8e5422c57348be65ccfab5625ec028ffd8c994d635e5a2","observation_id":"88ad8c4c-0f1e-420f-8a82-ee386e1cc1c4","resolution":{"observed_at":"2026-08-06T21:52:29.951040Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:29.738356Z","title":"UrbanMLLM: Joint learning of cross-view imagery for urban understanding, 2025","venue":null,"work_id":"9f1bd4a4-e91f-4d4a-b925-d42990e03bc2","year":2025},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:24.721919Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:4d82ad749fb11d2e79eec0e64fb88e9b8a0c9a6010725ac62a7a8ce206711654","observation_id":"f16343f6-e71c-4829-b6df-6ff60778e59a","resolution":{"observed_at":"2026-08-06T21:52:29.815763Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:29.536403Z","title":"Per- ceiving urban inequality from imagery using visual language models with chain-of-thought reasoning","venue":null,"work_id":"b8f9f4f6-e9ae-435d-8d26-84b21cd39c3e","year":2025},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:24.801393Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:f158ec234a04f88472de396e6cad7d305d915c8ee8e39a91ad05481dda6b948a","observation_id":"5f8b831f-e89b-4263-a0c4-c25564bd2048","resolution":{"observed_at":"2026-08-06T21:52:29.618477Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:29.357883Z","title":"Urbench: A comprehensive bench- mark for evaluating large multimodal models in multi-view urban scenarios","venue":null,"work_id":"6b73fee9-3b9d-467a-bd40-1145521b1f58","year":2025},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:24.854721Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:8a40bd225f199b49a6d602f019fa936eabc90149c0f318d551dc46be10bdd217","observation_id":"d2bfc31c-65af-4250-886b-738b385c421d","resolution":{"observed_at":"2026-08-06T21:52:29.440727Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:29.167299Z","title":"Deep learning for cross-domain data fu- sion in urban computing: Taxonomy, advances, and outlook","venue":null,"work_id":"fd3725aa-b8aa-4f87-8b6f-e7ac9ba25c00","year":2025},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:24.894192Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:25069f27328157223cf87b0706536b85962423bb488debf1fd03c7d7e0a78caf","observation_id":"29ccec06-f7c7-4493-9da2-98a5f2a90a87","resolution":{"observed_at":"2026-08-06T21:52:29.257763Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:28.889863Z","title":"Figure 9","venue":null,"work_id":"fd4cad17-6bf1-4599-9b9c-4b53e92ea55e","year":null},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:24.963116Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:8890896d458a7507d45fd9e1f32adf8165de0f0e0cbf54344bc313ed74e0adba","observation_id":"ef3ed754-043d-4ed6-bdfa-550a9d2985bb","resolution":{"observed_at":"2026-08-06T21:52:29.008529Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:28.620238Z","title":null,"venue":null,"work_id":"1531348f-0d09-48ce-bb8b-f5482f6cef6b","year":null},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:25.019923Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:9c493b6e4a888bb8f0d86d0e8080315a5586b6d2aa1cbff987dedca16891f384","observation_id":"f83a9c12-7d2c-48a3-a27f-632c538467da","resolution":{"observed_at":"2026-08-06T21:52:28.721388Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:28.408659Z","title":"Table 2 in Section 3.2 is the aggregated results of these three tables","venue":null,"work_id":"dafabbb8-ad15-49d2-a325-7ac807f9a564","year":null},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:25.067667Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:78290bab68401cc22fbc888b7e5325fffd894beeb7f29248f9d883e79aad516b","observation_id":"69414709-d8dc-4be6-ad26-cff32443ebda","resolution":{"observed_at":"2026-08-06T21:52:28.498738Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:28.115309Z","title":null,"venue":null,"work_id":"26dc772d-bb54-4073-a2a5-8798df7204a8","year":null},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:25.119209Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:3d49b07f49ce0847e452a513402bbea6c5413a1ebf1d062ca93c0cba7e64c92a","observation_id":"422b2167-4930-4227-b49b-cd6c734d3fad","resolution":{"observed_at":"2026-08-06T21:52:28.278570Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:27.832901Z","title":"11 presents training results with different amounts, ex- hibiting the high quality of UData","venue":null,"work_id":"67b65741-a8aa-4935-86a9-e0a8e41cf41b","year":null},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:25.203938Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:316ac047c4ffa64d697555fbe7599345ded4460bd61143ad09b966f8da56733d","observation_id":"fd295feb-c4b9-4d64-964b-22449050b58c","resolution":{"observed_at":"2026-08-06T21:52:27.972132Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:27.483020Z","title":null,"venue":null,"work_id":"8122fa6b-eae9-43fe-b49d-da66d5ffb0b2","year":null},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:25.253531Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:a51a824183d3f54a728834c282fe04cba8b9da54e2ee8a151bdeeccdc240aad3","observation_id":"8344fae6-6c4b-4b8f-a970-3f0b5373f535","resolution":{"observed_at":"2026-08-06T21:52:27.625332Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:27.250148Z","title":"However, for certain tasks, models of different sizes exhibit similar capabilities","venue":null,"work_id":"7819c4ce-de75-4c0f-9db6-9eb0c33b57ec","year":2000},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:25.315959Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:8abedd32f883281ec9ef18c18b46d40fdac8135d125bffae33e4d1acbcefec7f","observation_id":"7385252b-4d0d-42e6-8b30-6911e05c794a","resolution":{"observed_at":"2026-08-06T21:52:27.351086Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:26.944858Z","title":"This task needs a model to speculate the land use type (commercial, residential, agricultural, etc.) based on a satellite image","venue":null,"work_id":"c3be549d-ca24-4d0a-abcb-f8fbb23c9da2","year":1920},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:25.382008Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:b9096132de079af2a4e082e3507d0c45a3d0f024ea0726d3a087f8ef426cf59c","observation_id":"bd8fd18b-f630-4512-a500-92c471c491ba","resolution":{"observed_at":"2026-08-06T21:52:27.097941Z","resolver_source":"raw_fallback","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T21:52:26.524227Z","title":null,"venue":null,"work_id":"79a0fccf-2176-4b31-aa7f-e40c0807cdbe","year":null},"citing_paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-06T21:52:25.446679Z"},"links":{"citing_paper":"/paper/2506.23219"},"observation_digest":"sha256:fa3be1aa4683a60dac39a91f932471d9b499e03dcb50a51bcb595c5f8c1662e8","observation_id":"272f3bd7-3e1a-488c-b297-f5dd07e3e998","resolution":{"observed_at":"2026-08-06T21:52:26.762359Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.23219","last_updated":"2025-06-29T13:04:27Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-07T21:43:46.877530Z","submitted_at":"2025-06-29T13:04:27Z","title":"UrbanLLaVA: A Multi-modal Large Language Model for Urban Intelligence with Spatial Reasoning and Understanding"},"reference_resolution":{"displayed":66,"state_counts":{"malformed_identifier":2,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":28,"verified_exact":1,"verified_fuzzy":35},"total_outbound_references":66},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 66 of 66 outbound references and 2 inbound Pith citation observations for arXiv:2506.23219."}