{"as_of":"2026-08-23T20:35:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:49e1dc5d0b302d9b4dbb6f6db7b4a93acb7e80f9c0b899f1b07ae72229f4919e","coverage":[{"denominator":65,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":65,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T11:56:34.381892Z","state":"measured"},{"denominator":74,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":74,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-23T06:30:58.430688+00:00","state":"measured"},{"denominator":9,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":9,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-14T04:34:35.166153Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-02T02:16:27.244688Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"cited_work":{"arxiv_id":"2506.01031","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.01031","snapshot_observed_at":"2026-07-02T02:16:27.244688Z","title":"Navbench: Probing multimodal large language models for embodied navigation","venue":null,"work_id":"07e95eaf-1654-48cd-afae-50e9d01b313f","year":2025},"citing_paper":{"arxiv_id":"2602.18600","last_updated":"2026-07-29T09:42:45Z","snapshot_observed_at":"2026-08-02T22:00:02.664173Z","submitted_at":"2026-02-20T20:22:18Z","title":"MapTab: A Diagnostic Benchmark for Long-Horizon Multi-Criteria Multimodal Reasoning on Heterogeneous Topological Graphs","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-15T20:12:46.385646Z"},"links":{"cited_paper":"/paper/2506.01031","citing_paper":"/paper/2602.18600"},"observation_digest":"sha256:5604b769076a9d8041daf62231ba4cd2fb39e02cc0ba6779c5f2fd152dde9730","observation_id":"5a62e989-4c3d-4ef9-95d9-ea7cb53c7295","resolution":{"observed_at":"2026-05-15T20:16:34.582778Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"cited_work":{"arxiv_id":"2506.01031","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.01031","snapshot_observed_at":"2026-07-02T02:16:27.244688Z","title":"Navbench: Probing multimodal large language models for embodied navigation","venue":null,"work_id":"07e95eaf-1654-48cd-afae-50e9d01b313f","year":2025},"citing_paper":{"arxiv_id":"2602.18600","last_updated":"2026-07-29T09:42:45Z","snapshot_observed_at":"2026-08-02T22:00:02.664173Z","submitted_at":"2026-02-20T20:22:18Z","title":"MapTab: A Diagnostic Benchmark for Long-Horizon Multi-Criteria Multimodal Reasoning on Heterogeneous Topological Graphs","version":3},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-22T10:30:06.829915Z"},"links":{"cited_paper":"/paper/2506.01031","citing_paper":"/paper/2602.18600"},"observation_digest":"sha256:4b38cc468f5d2320a1c56bd016398619df9c3888d3736f10c5e5780bf786a036","observation_id":"8f94560f-6d13-458d-9dc2-364341ad59d5","resolution":{"observed_at":"2026-05-22T10:31:25.399311Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.01031","snapshot_observed_at":"2026-08-02T22:00:08.208686Z","title":"Navbench: Probing multimodal large language models for embodied navigation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2602.18600","last_updated":"2026-07-29T09:42:45Z","snapshot_observed_at":"2026-08-02T22:00:02.664173Z","submitted_at":"2026-02-20T20:22:18Z","title":"MapTab: A Diagnostic Benchmark for Long-Horizon Multi-Criteria Multimodal Reasoning on Heterogeneous Topological Graphs","version":5},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-02T22:00:08.208686Z"},"links":{"cited_paper":"/paper/2506.01031","citing_paper":"/paper/2602.18600"},"observation_digest":"sha256:872d8537d17385e292a310673ac93909d332eb031abc528f258c9bc58226565d","observation_id":"9e96874f-a3bc-4662-ac17-b6d6368fe404","resolution":{"observed_at":"2026-08-02T22:00:08.208686Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"cited_work":{"arxiv_id":"2506.01031","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.01031","snapshot_observed_at":"2026-07-02T02:16:27.244688Z","title":"Navbench: Probing multimodal large language models for embodied navigation","venue":null,"work_id":"07e95eaf-1654-48cd-afae-50e9d01b313f","year":2025},"citing_paper":{"arxiv_id":"2604.14785","last_updated":"2026-04-22T14:57:48Z","snapshot_observed_at":"2026-08-11T12:11:07.254950Z","submitted_at":"2026-04-16T08:45:34Z","title":"MirrorBench: Evaluating Self-centric Intelligence in MLLMs by Introducing a Mirror","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-10T10:53:48.374637Z"},"links":{"cited_paper":"/paper/2506.01031","citing_paper":"/paper/2604.14785"},"observation_digest":"sha256:f5eb3c127ede6946921fb48a5d569ab74abcf33aed808a364a9010d4e1e0e506","observation_id":"63e27b80-1468-4662-91e6-5979965a309b","resolution":{"observed_at":"2026-05-10T10:55:03.971070Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"cited_work":{"arxiv_id":"2506.01031","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.01031","snapshot_observed_at":"2026-07-02T02:16:27.244688Z","title":"Navbench: Probing multimodal large language models for embodied navigation","venue":null,"work_id":"07e95eaf-1654-48cd-afae-50e9d01b313f","year":2025},"citing_paper":{"arxiv_id":"2605.10118","last_updated":"2026-05-11T07:34:30Z","snapshot_observed_at":"2026-08-15T10:13:31.897860Z","submitted_at":"2026-05-11T07:34:30Z","title":"Plan in Sandbox, Navigate in Open Worlds: Learning Physics-Grounded Abstracted Experience for Embodied Navigation","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-05-12T03:36:24.941205Z"},"links":{"cited_paper":"/paper/2506.01031","citing_paper":"/paper/2605.10118"},"observation_digest":"sha256:449728af2c1c800700b0ed488f2d4da426f99f4eb376070c2615e18014d806cf","observation_id":"309f744d-0ad3-465b-94b4-4d66cfe3cf21","resolution":{"observed_at":"2026-05-12T07:11:27.618343Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"cited_work":{"arxiv_id":"2506.01031","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.01031","snapshot_observed_at":"2026-07-02T02:16:27.244688Z","title":"Navbench: Probing multimodal large language models for embodied navigation","venue":null,"work_id":"07e95eaf-1654-48cd-afae-50e9d01b313f","year":2025},"citing_paper":{"arxiv_id":"2605.15951","last_updated":"2026-05-15T13:41:41Z","snapshot_observed_at":"2026-08-16T20:40:52.226589Z","submitted_at":"2026-05-15T13:41:41Z","title":"From Failure to Feedback: Group Revision Unlocks Hard Cases in Object-Level Grounding","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-05-20T18:39:11.904941Z"},"links":{"cited_paper":"/paper/2506.01031","citing_paper":"/paper/2605.15951"},"observation_digest":"sha256:c7d6292c3fdc4a78cdf9f0159a7ec5bbbca37565861f39f4f255a5ddd8436597","observation_id":"1bcaed83-10ac-41c5-986b-486efaa7ddc1","resolution":{"observed_at":"2026-05-20T18:43:38.820539Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"cited_work":{"arxiv_id":"2506.01031","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.01031","snapshot_observed_at":"2026-07-02T02:16:27.244688Z","title":"Navbench: Probing multimodal large language models for embodied navigation","venue":null,"work_id":"07e95eaf-1654-48cd-afae-50e9d01b313f","year":2025},"citing_paper":{"arxiv_id":"2606.03175","last_updated":"2026-06-03T03:34:51Z","snapshot_observed_at":"2026-08-06T03:40:21.903218Z","submitted_at":"2026-06-02T05:31:03Z","title":"Ask When It Pays: Cost-Aware Open-Ended Interaction for Instance Goal Navigation","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-06-28T11:01:08.668612Z"},"links":{"cited_paper":"/paper/2506.01031","citing_paper":"/paper/2606.03175"},"observation_digest":"sha256:958e31b344e78e0ecf28273830af6c96c5df439b0c6a7c6947a62b1849100c3f","observation_id":"94f19bab-96e5-4f0a-9326-b856187046dd","resolution":{"observed_at":"2026-07-02T02:16:27.247643Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.01031","snapshot_observed_at":"2026-08-01T12:04:42.722556Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.19695","last_updated":"2026-07-22T02:53:46Z","snapshot_observed_at":"2026-08-21T02:54:10.537400Z","submitted_at":"2026-07-22T02:53:46Z","title":"NavVerse: Benchmarking Indoor-to-Outdoor Embodied Navigation in Continuous Robot Simulation","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-01T12:04:42.722556Z"},"links":{"cited_paper":"/paper/2506.01031","citing_paper":"/paper/2607.19695"},"observation_digest":"sha256:a540e1cdc322c2a640418b62f23c62c1be1eebf618ed471b4b293f17931d7ff9","observation_id":"046acf49-1f28-4c56-b115-bcc16b03c8cd","resolution":{"observed_at":"2026-08-01T12:04:42.722556Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.01031","snapshot_observed_at":"2026-08-14T04:34:35.166153Z","title":"arXiv preprint arXiv:2506.01031 (2025) 8","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.08596","last_updated":"2026-08-09T09:21:47Z","snapshot_observed_at":"2026-08-18T08:47:30.146792Z","submitted_at":"2026-08-09T09:21:47Z","title":"Goal-oriented Navigation Instruction Generation with Tour Video Priors","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-14T04:34:35.166153Z"},"links":{"cited_paper":"/paper/2506.01031","citing_paper":"/paper/2608.08596"},"observation_digest":"sha256:a96ab719e0084905fcd5fe788107e18c8b2d433d0ad50be77940ab9f98620793","observation_id":"81cd374f-3024-449e-aadb-4a19f7f1b7ae","resolution":{"observed_at":"2026-08-14T04:34:35.166153Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2506.01031/citation-record","integrity":"/paper/2506.01031/integrity","json":"/paper/2506.01031/citation-record.json","paper":"/paper/2506.01031"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:30.989880Z","title":"Visual instruction tuning","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:30.989880Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:8587bfb0046007052cc06c157cd26b61b9816946c1fb9813e2bc6228a53f4463","observation_id":"21215b46-a989-48d6-a31a-5c189b1c4039","resolution":{"observed_at":"2026-08-07T11:56:30.989880Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.12191","last_updated":"2024-10-03T15:54:49Z","snapshot_observed_at":"2026-08-06T05:35:29.109022Z","submitted_at":"2024-09-18T17:59:32Z","title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.12191","snapshot_observed_at":"2026-08-07T11:56:31.036599Z","title":"Qwen2-vl: Enhancing vision-language model’s perception of the world at any resolution","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:31.036599Z"},"links":{"cited_paper":"/paper/2409.12191","citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:692f4dd62496dc0959805bf4a56978f965d4fe3ba7eba836af29f249a01de214","observation_id":"d6dcc617-bf87-405b-8a38-f2ac8f5e3a98","resolution":{"observed_at":"2026-08-07T11:56:31.036599Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05530","last_updated":"2024-12-16T17:39:39Z","snapshot_observed_at":"2026-08-14T18:15:53.516440Z","submitted_at":"2024-03-08T18:54:20Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.05530","snapshot_observed_at":"2026-08-07T11:56:31.156886Z","title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:31.156886Z"},"links":{"cited_paper":"/paper/2403.05530","citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:9df42ffe3e51cbefce3f32759fd11faaad71e7bba82cb725933c64f7df4b52be","observation_id":"7bf7b956-907d-44c0-83bb-03a43bee0878","resolution":{"observed_at":"2026-08-07T11:56:31.156886Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:31.269579Z","title":"Making the v in vqa matter: Elevating the role of image understanding in visual question answering","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:31.269579Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:4a9ef5458365239e8dca593ffbfbcdfc99634ca5430f884f58f2d5dbafec84b7","observation_id":"c51a4e45-7c00-4f36-9154-1f413d56c8c6","resolution":{"observed_at":"2026-08-07T11:56:31.269579Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.21075","last_updated":"2025-05-30T13:08:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-05-31T17:59:47Z","title":"Video-MME: The First-Ever Comprehensive Evaluation Benchmark of Multi-modal LLMs in Video Analysis","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.21075","snapshot_observed_at":"2026-08-07T11:56:31.360091Z","title":"Video-mme: The first-ever comprehensive evaluation benchmark of multi-modal llms in video analysis","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:31.360091Z"},"links":{"cited_paper":"/paper/2405.21075","citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:f11f3896a43a7a8a69c0932a85676858f5bf9975d20d0a2f29f35e46ff690036","observation_id":"c4c0c367-e3e5-464f-a8e7-002e1df9fd07","resolution":{"observed_at":"2026-08-07T11:56:31.360091Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.02255","last_updated":"2024-01-21T03:47:06Z","snapshot_observed_at":"2026-08-20T14:58:44.877503Z","submitted_at":"2023-10-03T17:57:24Z","title":"MathVista: Evaluating Mathematical Reasoning of Foundation Models in Visual Contexts","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.02255","snapshot_observed_at":"2026-08-07T11:56:31.435423Z","title":"Mathvista: Evaluating mathematical reasoning of foundation models in visual contexts","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:31.435423Z"},"links":{"cited_paper":"/paper/2310.02255","citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:55c6990ec8d8c1f5d6a47987c0d1ccb2e911681436e75257bf906f0a2be451ea","observation_id":"ada2f096-3cec-4aec-97fd-2809ac0113fc","resolution":{"observed_at":"2026-08-07T11:56:31.435423Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.837151Z","title":"Spatialbot: Precise spatial understanding with vision language models","venue":null,"work_id":"9db9fd41-dbc6-45c6-ac2a-f83d8f272d9f","year":2025},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:31.538043Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:e77c3d9cf7fd4b36b7ce6b62504553c35fce5e745b0065908fa77a55113d8e01","observation_id":"7b4c1524-e616-4184-8a8f-eaa4f92420c5","resolution":{"observed_at":"2026-08-07T11:56:34.839389Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.830581Z","title":"Thinking in Space: How Multimodal Large Language Models See, Remember and Recall Spaces","venue":null,"work_id":"98e8f3ab-3253-440f-9cf4-99fff780d204","year":2025},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:31.656410Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:ddb55d075ccd7a6fe974bb293855002c45c9e4ed8c30be93e95a12ab355d7ce3","observation_id":"a0924245-280f-45f5-899b-f0f7bd9a23f3","resolution":{"observed_at":"2026-08-07T11:56:34.833229Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.824438Z","title":"Reid, Stephen Gould, and Anton van den Hengel","venue":null,"work_id":"2d5361e4-9e42-4cfc-b052-0496ce6bcb98","year":2018},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:31.740055Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:07556e5c705454d9753d1273110f75bc5e984832175193ef50a3554ebb7bb4e3","observation_id":"030c5f51-3ec7-4655-9446-10e0009e8a9b","resolution":{"observed_at":"2026-08-07T11:56:34.826606Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:31.853061Z","title":"Object goal navigation using goal-oriented semantic exploration","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:31.853061Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:09116d39abf15677aaeb4d5531c77d33d5c00a639e65295ec37dba2623baf478","observation_id":"9bd105d3-efdd-45e2-b6d8-8310cd3906f0","resolution":{"observed_at":"2026-08-07T11:56:31.853061Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.813751Z","title":"The spatial semantic hierarchy","venue":null,"work_id":"b70bfacc-c3bf-45fc-94af-ffe406229d06","year":2000},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:31.993816Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:a1a2a047f9598ccb0ca708519d8badd6f2c0b045ad86fb4e6cc074dc776097a1","observation_id":"163a60bc-f90e-4919-971f-bc8a1bf2d7fa","resolution":{"observed_at":"2026-08-07T11:56:34.815799Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.806856Z","title":"Intern vl: Scaling up vision foundation models and aligning for generic visual-linguistic tasks","venue":null,"work_id":"1835a232-dfd8-49ed-8bc6-5f2d60d0d5fb","year":2023},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:32.094607Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:1760f9c506f9f8a78ec52133a94b3518290f04ebb4e3d216495f0ca8b29b389c","observation_id":"956b1190-8d93-4d6e-b833-e2abbe250619","resolution":{"observed_at":"2026-08-07T11:56:34.809216Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2204.14198","last_updated":"2022-11-15T23:07:37Z","snapshot_observed_at":"2026-08-17T03:00:38.806206Z","submitted_at":"2022-04-29T16:29:01Z","title":"Flamingo: a Visual Language Model for Few-Shot Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2204.14198","snapshot_observed_at":"2026-08-07T11:56:32.233009Z","title":"Flamingo: a visual language model for few-shot learning","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:32.233009Z"},"links":{"cited_paper":"/paper/2204.14198","citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:1e20b880d3fa4f4d47945efe4a79ec6e64e18280c7bf12c3e72bb604a6602d21","observation_id":"43a30cec-1dad-4cb1-9574-8989f357a51c","resolution":{"observed_at":"2026-08-07T11:56:32.233009Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.12966","last_updated":"2023-10-13T02:41:28Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-08-24T17:59:17Z","title":"Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.12966","snapshot_observed_at":"2026-08-07T11:56:32.381639Z","title":"Qwen-vl: A frontier large vision-language model with versatile abilities","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:32.381639Z"},"links":{"cited_paper":"/paper/2308.12966","citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:ba2201561a08936b487bc8167c46e62cd3c21283cfdb991ef6d666d85540ad39","observation_id":"1ae2745b-9f2c-4dde-a16d-bd98599fb154","resolution":{"observed_at":"2026-08-07T11:56:32.381639Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.800180Z","title":"Improved baselines with visual instruction tuning","venue":null,"work_id":"35af17f0-26f3-4464-b1a1-da1150d348be","year":2023},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:32.448570Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:d080fe4a1c145379ef21db8934b4e8971f3845abc5c24437ab0c9de9b656ad04","observation_id":"50155778-7ad6-4327-93f5-5d026fd1ce09","resolution":{"observed_at":"2026-08-07T11:56:34.802388Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:32.535038Z","title":"Gqa: A new dataset for real-world visual reasoning and compositional question answering","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:32.535038Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:e5fb48cd40ca4905ea8ed981994176ca1394b4ae150ccff22308bc588906ce55","observation_id":"4c307aed-a278-4a95-be65-6c1593812cc2","resolution":{"observed_at":"2026-08-07T11:56:32.535038Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:32.613705Z","title":"Ok-vqa: A visual question answering benchmark requiring external knowledge","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:32.613705Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:1b2cfb6188272637427d14467edb16928a74c2fdf99b1e77fa0ae7771d9e3b56","observation_id":"666f5b79-3766-45a8-a37d-1c3bb08eef14","resolution":{"observed_at":"2026-08-07T11:56:32.613705Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:32.700436Z","title":"Towards vqa models that can read","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:32.700436Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:1da7590ec05ba820768a899e6c183c1996c44a17e563a1fd4613c38684dff104","observation_id":"0e8687be-74f3-4bb6-970e-1e0b67838695","resolution":{"observed_at":"2026-08-07T11:56:32.700436Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.13549","last_updated":"2024-11-29T15:51:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-23T15:21:52Z","title":"A Survey on Multimodal Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.13549","snapshot_observed_at":"2026-08-07T11:56:32.778551Z","title":"A survey on multimodal large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:32.778551Z"},"links":{"cited_paper":"/paper/2306.13549","citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:710ac460186d0f1cd5bde6d34404c027400759751258536d38cffaa065dcf3af","observation_id":"7a65ad5c-e12b-4aa9-9f5e-735fd19daff0","resolution":{"observed_at":"2026-08-07T11:56:32.778551Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:32.876652Z","title":"Mmbench: Is your multi-modal model an all-around player? In European Conference on Computer Vision, pages 216–233","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:32.876652Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:d94c682dc035bf415f276dc2075de652039eb09b57f47239b94cb8925cc3d634","observation_id":"50ae15e1-5e0c-4f5b-9528-576352a5248b","resolution":{"observed_at":"2026-08-07T11:56:32.876652Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.02490","last_updated":"2024-12-01T05:46:03Z","snapshot_observed_at":"2026-08-21T04:07:07.949331Z","submitted_at":"2023-08-04T17:59:47Z","title":"MM-Vet: Evaluating Large Multimodal Models for Integrated Capabilities","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.02490","snapshot_observed_at":"2026-08-07T11:56:32.992111Z","title":"Mm-vet: Evaluating large multimodal models for integrated capabilities","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:32.992111Z"},"links":{"cited_paper":"/paper/2308.02490","citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:ea55476b91ed784f54fccd127e85bc3a73c7deb34aba6f5a4c0c269ff534a576","observation_id":"e4bf5719-35c4-4031-9ea6-a8bb600ab2c0","resolution":{"observed_at":"2026-08-07T11:56:32.992111Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.713801Z","title":"Scanreason: Empowering 3d visual grounding with reasoning capabilities","venue":null,"work_id":"7c5fe45b-78c2-4b69-832e-0c8d2d56e9de","year":2024},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:33.138520Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:ba1bef3970f46fcf7bc81b8a451fa15728e567c0e968d797bb683cbc4f5ac518","observation_id":"a39b7188-b251-4424-80d0-02ee0fbfd917","resolution":{"observed_at":"2026-08-07T11:56:34.779899Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.708006Z","title":"REVERIE: remote embodied visual referring expression in real indoor environments","venue":null,"work_id":"7040ca00-3cc1-4982-b30a-5208aadf9afe","year":2020},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:33.218505Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:40ce122ac35f63d472af8ef2408cd21126d75a89b840d5138f4d21870025472e","observation_id":"0ee7bb89-df2f-4513-9e83-3155b4816b88","resolution":{"observed_at":"2026-08-07T11:56:34.710186Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.17313","last_updated":"2024-09-25T19:49:39Z","snapshot_observed_at":"2026-08-20T09:35:38.469696Z","submitted_at":"2024-09-25T19:49:39Z","title":"Navigating the Nuances: A Fine-grained Evaluation of Vision-Language Navigation","version":1},"cited_work":{"arxiv_id":"2409.17313","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.17313","snapshot_observed_at":"2026-08-07T11:56:34.456983Z","title":"Navigating the Nuances: A Fine-grained Evaluation of Vision-Language Navigation","venue":"cs.CV","work_id":"9f32ad88-995c-4e5a-b186-c6d2f6684b14","year":2024},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:33.316054Z"},"links":{"cited_paper":"/paper/2409.17313","citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:70337042d240101886d615c91527b2b58cc1b251902bce6f9ddb0057e44aa628","observation_id":"9ab1db93-f232-4bd4-aa0f-7d98aaeb8838","resolution":{"observed_at":"2026-08-07T11:56:34.461252Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.701691Z","title":"Target-driven visual navigation in indoor scenes using deep reinforcement learning","venue":null,"work_id":"9c2204b7-b1b3-47e7-b539-07391ef71fd0","year":2017},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:33.442990Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:d0c736b732657c0643ee7c2d785696cebbba7c06fcbdd347fdd4ff74bc011563","observation_id":"b7fa6440-7214-4ee8-a7fc-9be78f94c935","resolution":{"observed_at":"2026-08-07T11:56:34.704008Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2211.15876","last_updated":"2022-11-29T02:29:35Z","snapshot_observed_at":"2026-08-20T10:17:32.122864Z","submitted_at":"2022-11-29T02:29:35Z","title":"Instance-Specific Image Goal Navigation: Training Embodied Agents to Find Object Instances","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2211.15876","snapshot_observed_at":"2026-08-07T11:56:33.633441Z","title":"Instance-specific image goal navigation: Training embodied agents to find object instances.arXiv preprint arXiv:2211.15876, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:33.633441Z"},"links":{"cited_paper":"/paper/2211.15876","citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:63495409a87bd1933490e25986cacdc72551580674a2635e834224d08b5eba67","observation_id":"0a07806e-9fd6-482a-a798-a6ea9258cc60","resolution":{"observed_at":"2026-08-07T11:56:33.633441Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.695382Z","title":"Habitat: A platform for embodied ai research","venue":null,"work_id":"0f41d3b9-9377-4fc4-89f6-bcd543745899","year":2019},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:33.786563Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:8530de54f40a1d2002924374c6b49f1cd2e25813cddfc478f780ba5fe9798210","observation_id":"d0a96ec0-53bb-487f-8be0-5d67cb8cc933","resolution":{"observed_at":"2026-08-07T11:56:34.697575Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.688903Z","title":"History aware multimodal transformer for vision-and-language navigation","venue":null,"work_id":"85a30438-7c16-42d2-9985-bf6d9e3d045e","year":2021},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:33.926155Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:77e82579517bd3bdccd2c07c9971f651a4b667f14f6b545963abe19f89a628e0","observation_id":"d28e1034-482c-4f62-ac0d-51f97579a780","resolution":{"observed_at":"2026-08-07T11:56:34.691481Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.682921Z","title":"Vision-and-language navigation today and tomorrow: A survey in the era of foundation models","venue":null,"work_id":"7048cd78-9a98-4a38-b9f3-a4fab928a97a","year":2024},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:33.931292Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:3c7e0e6e53a74852ba72f4421cce03a5e5fc01cbc106a025abd8ab678b211449","observation_id":"ffb1cc7d-e6cb-40f0-9866-b62c3758316a","resolution":{"observed_at":"2026-08-07T11:56:34.685086Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.676086Z","title":"Room-across-room: Mul- tilingual vision-and-language navigation with dense spatiotemporal grounding","venue":null,"work_id":"421b4cff-b534-4441-8e12-1239c326430c","year":2020},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.016628Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:8e29b8ab4065d6996befbdeed328c6adacdfee73b42c2eb172c0398a7c3d2a74","observation_id":"ff5f2acb-1254-47a2-8950-80c4aa765228","resolution":{"observed_at":"2026-08-07T11:56:34.678492Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.669850Z","title":"Vision-and-dialog navigation","venue":null,"work_id":"4a34bd30-d508-46b5-bb8c-c63aff41fc34","year":2019},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.129147Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:8f7cf60ed1073d2f3356c98a977b7f0eef8043517a9aa1be639c1c0946fe88e5","observation_id":"da9e8b29-1c88-4862-a82f-6446963e2b00","resolution":{"observed_at":"2026-08-07T11:56:34.671826Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.663625Z","title":"Find what you want: Learning demand-conditioned object attribute space for demand-driven navigation","venue":null,"work_id":"0b50992f-7ac0-4f43-a39b-4b6b00498201","year":2023},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.239516Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:26d8c85c765ac7af4b0ee74de9e6e6aeba785f5da2075bbcd9e98bbb2ddd0fb7","observation_id":"3b45aee4-bb4f-48f3-b4ff-f721f41f20bb","resolution":{"observed_at":"2026-08-07T11:56:34.665891Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.657169Z","title":"Object-and-action aware model for visual language navigation","venue":null,"work_id":"d733b932-b474-446e-87e6-7c6487bae049","year":2020},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.302998Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:3c89fb3720fd449624477e6d57dc6851600f2e55e64724a6a12fbac12008e618","observation_id":"dbd5fa5b-0761-4392-9c2a-2798e5b23685","resolution":{"observed_at":"2026-08-07T11:56:34.659387Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.650178Z","title":"Language and visual entity relationship graph for agent navigation","venue":null,"work_id":"f9568e84-e00a-42b3-a665-ba840d87cab4","year":2020},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.305366Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:3ea47f0d4c016080e2c43756b59db397177edca5605c9cfe8dd1aa5c8cfedb1c","observation_id":"6fdfc8e7-7c36-4f64-b176-5eac9219a851","resolution":{"observed_at":"2026-08-07T11:56:34.652728Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.643410Z","title":"Neighbor-view enhanced model for vision and language navigation","venue":null,"work_id":"f15d818a-16fd-4808-a5e2-d099776c0c0b","year":2021},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.308363Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:b9b87f9973ccdd13d9621c7990ee1de1465491fc15c712d2cceb15f4cda0c0b5","observation_id":"ffe8f20b-1dee-48f8-8931-bdb2673dad8d","resolution":{"observed_at":"2026-08-07T11:56:34.645702Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.311272Z","title":"Towards learning a generic agent for vision-and-language navigation via pre-training","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.311272Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:df316ae1d6f0ca44a9a3da52cc2de3d7b69b7b69d99f7e8610c49a3a6107c06b","observation_id":"d6d414e9-6abf-4031-be40-b86b044e4d54","resolution":{"observed_at":"2026-08-07T11:56:34.311272Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.633314Z","title":"Airbert: In- domain pretraining for vision-and-language navigation","venue":null,"work_id":"d1538405-d9fd-453e-a171-b80b88dd4b58","year":2021},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.313903Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:38f60154c16a7a4317891fff94ce6f4e7808cafcd77c0d5c7791cb40699291eb","observation_id":"eb74ac20-a0b7-459f-90f3-1de2bbe0d4b7","resolution":{"observed_at":"2026-08-07T11:56:34.635480Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.626559Z","title":"Improving vision-and-language navigation with image-text pairs from the web","venue":null,"work_id":"54de2445-65dc-4d9a-9683-8af36d14e879","year":2020},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.316051Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:761169b662cee2906e0b3b99ea30930746b946ac749c3b50f5ad6620c063f3bd","observation_id":"797d208f-725c-4563-9f38-ad6d6a2eb10a","resolution":{"observed_at":"2026-08-07T11:56:34.628863Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.318070Z","title":"History aware multimodal transformer for vision-and-language navigation","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.318070Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:c9e8df4975553e20bd4fdb8f2033366c4fe3476ff415f9c45a41dd5bc648292e","observation_id":"f4007912-7675-4df8-ade3-55b576e94169","resolution":{"observed_at":"2026-08-07T11:56:34.318070Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.616452Z","title":"Hop: History-and-order aware pre-training for vision-and-language navigation","venue":null,"work_id":"055d3728-d752-4f7d-add3-a0621fcc6d20","year":2022},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.320729Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:cf54cb438f8c8cfbed4194db6e4e9337fefa1230fb8feefad1521908d7490d4f","observation_id":"e69bee68-35b1-427e-b8ea-e294a8136d1c","resolution":{"observed_at":"2026-08-07T11:56:34.619052Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.609981Z","title":"Hop+: History-enhanced and order-aware pre-training for vision-and-language navigation","venue":null,"work_id":"ef2843f3-692b-4789-9b82-4a8a4cf600da","year":2023},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.322616Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:6251532911c595e20136a3802431824b71d1bdfdfed77566531a5b4157b34d19","observation_id":"0cc9a409-1da3-46a0-b1c3-e5cb4312d527","resolution":{"observed_at":"2026-08-07T11:56:34.612135Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.602954Z","title":"Bevbert: Mul- timodal map pre-training for language-guided navigation","venue":null,"work_id":"fea186dd-697b-4f51-9de9-91a24f6d6e61","year":2023},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.324822Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:9e791fa846c5b3895cf543a2cf8c84015f2ac0f629eb836cc7f127c9ecbe6e25","observation_id":"cab423a7-5761-4eff-9ef3-0c661c24d1d2","resolution":{"observed_at":"2026-08-07T11:56:34.605762Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.596248Z","title":"Bird’s-eye-view scene graph for vision-language navigation","venue":null,"work_id":"3ac8b239-8e37-4982-8190-9e34d72047af","year":2023},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.327508Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:2a53cd8f749623f50acbf89af3e63338392fff9ea408210b7ebe526d256ace7a","observation_id":"ab3890ed-ae3b-4910-b9a6-8a119a827766","resolution":{"observed_at":"2026-08-07T11:56:34.598706Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.329612Z","title":"Gridmm: Grid memory map for vision-and-language navigation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.329612Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:fb87ecb100ceb15b3e366ff804f5e62b76810e94051de6b6683f3e2d59f74435","observation_id":"869c5aa6-ead2-4a95-b715-14fba6052152","resolution":{"observed_at":"2026-08-07T11:56:34.329612Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.585863Z","title":"Scaling data generation in vision-and-language navigation","venue":null,"work_id":"5fe28ed6-0e53-4b51-92b1-80739126d441","year":2023},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.332220Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:f36060fa47d6a30d8928d3522263733f0e0334df9247311e390ebfce8a93b593","observation_id":"b4148ee3-1211-4057-a773-47d20b81f475","resolution":{"observed_at":"2026-08-07T11:56:34.588372Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.578520Z","title":"A new path: Scaling vision-and-language navigation with synthetic instructions and imitation learning","venue":null,"work_id":"f67cd51b-1257-4a1d-b273-4f74800f157d","year":2023},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.334452Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:cc5fda9b77409959980c80a6a083a620bb58b115fdac0a4509c3a4019df9616c","observation_id":"fbd31e1a-43f8-4c87-8d75-681bf43297a7","resolution":{"observed_at":"2026-08-07T11:56:34.581922Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.15852","last_updated":"2024-06-30T11:14:13Z","snapshot_observed_at":"2026-08-18T12:12:42.503422Z","submitted_at":"2024-02-24T16:39:16Z","title":"NaVid: Video-based VLM Plans the Next Step for Vision-and-Language Navigation","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.15852","snapshot_observed_at":"2026-08-07T11:56:34.336469Z","title":"Navid: Video-based vlm plans the next step for vision-and-language navigation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.336469Z"},"links":{"cited_paper":"/paper/2402.15852","citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:9543de37bd009d038b37a09acef730f1c4187d64fa2aaf5fa27caa9ccc53fc12","observation_id":"fd583f3a-7b3f-48da-b60c-cef11520fcf8","resolution":{"observed_at":"2026-08-07T11:56:34.336469Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.338831Z","title":"Esc: Exploration with soft commonsense constraints for zero-shot object navigation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.338831Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:1425ba53ca7342e6990be6c2fbad3a164ff3652dab6014738949c746b4cbaa60","observation_id":"4c5f0d30-1b6d-4dce-952a-b5db2d52d47d","resolution":{"observed_at":"2026-08-07T11:56:34.338831Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.568607Z","title":"Cows on pasture: Baselines and benchmarks for language-driven zero-shot object navigation","venue":null,"work_id":"999dc60f-2d17-46e4-b68c-0e42c9f9f820","year":2023},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.341023Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:780efa25701fa6660f231e80d9f48e3ba75f2aa6692392a3415342fd38dc2f33","observation_id":"4c23622b-1486-448a-8ddf-6ff00fd9303f","resolution":{"observed_at":"2026-08-07T11:56:34.570863Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.562415Z","title":"Vlfm: Vision- language frontier maps for zero-shot semantic navigation","venue":null,"work_id":"cb521529-0a73-40ca-85a7-a78be4be1071","year":2024},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.343415Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:1087dc9380a1d076af9b1134dc82728e598b6afc1a0edef727d9cd8de0b0b366","observation_id":"13a87f84-42db-4e4b-924c-f64727b4c4f3","resolution":{"observed_at":"2026-08-07T11:56:34.564674Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04882","last_updated":"2024-06-07T12:26:34Z","snapshot_observed_at":"2026-08-16T13:45:10.310910Z","submitted_at":"2024-06-07T12:26:34Z","title":"InstructNav: Zero-shot System for Generic Instruction Navigation in Unexplored Environment","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04882","snapshot_observed_at":"2026-08-07T11:56:34.345807Z","title":"Instructnav: Zero-shot system for generic instruction navigation in unexplored environment","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.345807Z"},"links":{"cited_paper":"/paper/2406.04882","citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:c073606bb88fc799783aff8dc997be78d27f602b44e5dce71d98b58d59003295","observation_id":"f24c66b4-811a-44a7-b80f-3dca4d9ed00a","resolution":{"observed_at":"2026-08-07T11:56:34.345807Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.555790Z","title":"Navgpt: Explicit reasoning in vision-and-language navigation with large language models","venue":null,"work_id":"06680d85-e6e7-4ca1-9e73-89d07e0f6cd1","year":2024},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.348198Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:bb32243b5d11908d84de0ab7dfb1557a740e334af771043b986817f706e4eac1","observation_id":"e0db8a61-4362-467e-a5e9-b777b260f7d7","resolution":{"observed_at":"2026-08-07T11:56:34.558219Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.18794","last_updated":"2025-02-11T00:55:35Z","snapshot_observed_at":"2026-08-17T12:14:18.185480Z","submitted_at":"2024-09-27T14:47:18Z","title":"Open-Nav: Exploring Zero-Shot Vision-and-Language Navigation in Continuous Environment with Open-Source LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.18794","snapshot_observed_at":"2026-08-07T11:56:34.350222Z","title":"Open- nav: Exploring zero-shot vision-and-language navigation in continuous environment with open-source llms","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.350222Z"},"links":{"cited_paper":"/paper/2409.18794","citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:6575dcbde751c9de5145710d01c917a226f322164a6924568987988539f7bdbc","observation_id":"73da22ea-1bd2-4d23-a7d2-299ecd103800","resolution":{"observed_at":"2026-08-07T11:56:34.350222Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.549013Z","title":"Mapgpt: Map- guided prompting with adaptive path planning for vision-and-language navigation","venue":null,"work_id":"287b0994-f279-45a9-82b8-efcd76362869","year":2024},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.352853Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:b1152482cd982564f6a1634223e65ac35cabcbcd98fcc65d96aed9d4c8d6b2f1","observation_id":"ad13d19e-961a-4947-9563-355054a51117","resolution":{"observed_at":"2026-08-07T11:56:34.551288Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.542835Z","title":"Chang, Angela Dai, Thomas A","venue":null,"work_id":"40a66127-2a99-4a62-9133-27a210f0740c","year":2017},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.355084Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:fbc9f40606b46fcff3cf73575fc8486dd1b257b3b695266698474cb0734a6094","observation_id":"6edd141a-afbb-4880-acea-c47a3dc02043","resolution":{"observed_at":"2026-08-07T11:56:34.544972Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.536096Z","title":"Grounded entity-landmark adaptive pre-training for vision-and-language navigation","venue":null,"work_id":"b6dd2d50-c139-44d6-a1fb-b4f91d7acab6","year":2023},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.357892Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:af02bf3a8f628de0ba8263258543d8648074930eef85e429b6069450426e5507","observation_id":"e159f15b-d07c-49cd-8886-5a0df644850e","resolution":{"observed_at":"2026-08-07T11:56:34.538432Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.529789Z","title":"Sub-instruction aware vision-and-language navigation","venue":null,"work_id":"2fc8d8ee-b955-4bf1-8512-b32b8a4b27bf","year":2020},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.360525Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:eb9619d0ef1f3596cb4911310a37806e3fc23c1c4ac8e95be6a14e966aa10519","observation_id":"8849c3ca-7649-4007-87cb-a264cde8c80f","resolution":{"observed_at":"2026-08-07T11:56:34.531915Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.522442Z","title":"Are large vision language models good game players? In Proceedings of the International Conference on Learning Representations, 2025","venue":null,"work_id":"37f6d3fd-051d-432e-ace6-6e9dcafb875f","year":2025},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.362980Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:fa5c7da40117e1fecff96ded3d891ac2085efdb4a41330e02fbe1b1fc6821662","observation_id":"1a66783c-cad6-41fa-a3f5-806567c558ea","resolution":{"observed_at":"2026-08-07T11:56:34.525823Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.03978","last_updated":"2024-10-31T08:11:04Z","snapshot_observed_at":"2026-08-21T21:20:56.192266Z","submitted_at":"2024-07-04T14:50:45Z","title":"Benchmarking Complex Instruction-Following with Multiple Constraints Composition","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.03978","snapshot_observed_at":"2026-08-07T11:56:34.365773Z","title":"Benchmarking complex instruction-following with multiple constraints composition","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.365773Z"},"links":{"cited_paper":"/paper/2407.03978","citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:377ced9858ce04acc3236ed76012e96346524b141ac9a69bb3021b1ab270fd8b","observation_id":"06ac8df7-1398-4aa9-98f8-72ced6995343","resolution":{"observed_at":"2026-08-07T11:56:34.365773Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.05271","last_updated":"2025-09-26T12:52:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-06T18:57:08Z","title":"Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.05271","snapshot_observed_at":"2026-08-07T11:56:34.368716Z","title":"Expanding performance boundaries of open-source multimodal models with model, data, and test-time scaling","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.368716Z"},"links":{"cited_paper":"/paper/2412.05271","citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:1044cb2c755eadfc6430c4c4a754fd405443cc63a91af713d9495eb4478c1529","observation_id":"c1723077-1fe6-4428-b60f-2d9391758425","resolution":{"observed_at":"2026-08-07T11:56:34.368716Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.13923","last_updated":"2025-02-19T18:00:14Z","snapshot_observed_at":"2026-08-21T16:02:41.546193Z","submitted_at":"2025-02-19T18:00:14Z","title":"Qwen2.5-VL Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.13923","snapshot_observed_at":"2026-08-07T11:56:34.371350Z","title":"Qwen2.5-vl technical report","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.371350Z"},"links":{"cited_paper":"/paper/2502.13923","citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:73712168f5e2d8f8c0422d60a7e8260fed88c077bd4d4ce36eadd2df94027cb5","observation_id":"c024a099-385d-4309-b3a3-2bd1e3852bad","resolution":{"observed_at":"2026-08-07T11:56:34.371350Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.03326","last_updated":"2024-10-26T16:35:13Z","snapshot_observed_at":"2026-08-22T00:56:08.009523Z","submitted_at":"2024-08-06T17:59:44Z","title":"LLaVA-OneVision: Easy Visual Task Transfer","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.03326","snapshot_observed_at":"2026-08-07T11:56:34.374202Z","title":"Llava-onevision: Easy visual task transfer","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.374202Z"},"links":{"cited_paper":"/paper/2408.03326","citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:8dc018da6f91985a0af4f20d8db1ebcbd957060ff25ef463ef626ca9bdc9ba54","observation_id":"88b45deb-1d22-4ed4-9504-36f2cf5eab46","resolution":{"observed_at":"2026-08-07T11:56:34.374202Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-08-13T17:20:44.002518Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-07T11:56:34.377086Z","title":"The llama 3 herd of models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.377086Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:9f611e6c0740c6360f8c988ad3b197d1a742e7b7178bb0d3755b363e78fabefe","observation_id":"01e28082-e430-4a01-8720-f98fd4e5c1a1","resolution":{"observed_at":"2026-08-07T11:56:34.377086Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.379660Z","title":"Gonzalez, Hao Zhang, and Ion Stoica","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.379660Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:883c7253015b484dbf90667062cbcbcdd1b3a853791f46cc3e7de51964d1341e","observation_id":"e7a884cf-2ce2-4223-8b8d-c19b043549b3","resolution":{"observed_at":"2026-08-07T11:56:34.379660Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T11:56:34.511385Z","title":"Lmdeploy: A toolkit for compressing, deploying, and serving llm","venue":null,"work_id":"35cefa36-7cc2-4462-b348-6493c760e337","year":2023},"citing_paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:34.381892Z"},"links":{"citing_paper":"/paper/2506.01031"},"observation_digest":"sha256:e1dc5ae655a3eadf66c64326fd7eef0af8027837bdf1e362504ab9e9682a694a","observation_id":"71935fba-e03a-4d0a-b192-16795e431d8c","resolution":{"observed_at":"2026-08-07T11:56:34.513685Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.01031","last_updated":"2025-06-01T14:21:02Z","latest_version":1,"primary_category":"cs.CV","snapshot_observed_at":"2026-08-13T21:47:20.221397Z","submitted_at":"2025-06-01T14:21:02Z","title":"NavBench: Probing Multimodal Large Language Models for Embodied Navigation"},"reference_resolution":{"displayed":65,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":29,"verified_exact":1,"verified_fuzzy":35},"total_outbound_references":65},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"thesis":"As of 23 August 2026, this Paper Citation Record lists 65 of 65 outbound references and 9 inbound Pith citation observations for arXiv:2506.01031."}