{"total":19,"items":[{"citing_arxiv_id":"2607.05765","ref_index":97,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Image2Sim: Scaling Embodied Navigation via Generative Neural Simulator","primary_cat":"cs.CV","submitted_at":"2026-07-07T02:42:41+00:00","verdict":"CONDITIONAL","verdict_confidence":"HIGH","novelty_score":6.5,"formal_verification":"none","one_line_summary":"A feed-forward feature-Gaussian plus one-step geometry-aware pixel-flow simulator converts large image collections into 20K interactive scenes and 10M+ navigation samples that improve zero-shot Habitat and real-robot performance.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.05377","ref_index":23,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Cortex: A Bidirectionally Aligned Embodied Agent Framework for Long-horizon Manipulation","primary_cat":"cs.RO","submitted_at":"2026-07-06T17:55:05+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":5.0,"formal_verification":"none","one_line_summary":"A dual-system framework with a structured subtask interface, event-balanced training, and inference harness enables VLM-guided long-horizon robotic manipulation, achieving 95.5% on LIBERO-Long and 65% on real-world chemistry tasks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.02417","ref_index":33,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"LIME: Learning Intent-aware Camera Motion from Egocentric Video","primary_cat":"cs.RO","submitted_at":"2026-07-02T16:48:43+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"LIME formulates language-conditioned camera motion as predicting SE(3) target poses from RGB and intent text, using mined multi-intent supervision from egocentric video and a flow-matching pose head.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.30367","ref_index":76,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"FutureNav: Unified World-Action Modeling for Vision-and-Language Navigation","primary_cat":"cs.RO","submitted_at":"2026-06-29T14:33:03+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"FutureNav proposes a 4B-scale VLM that jointly optimizes action prediction, inverse/forward dynamics, and future state generation for VLN and reports SOTA results on multiple benchmarks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.20458","ref_index":15,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Slow Brain, Fast Planner: Latency-Resilient VLM-Augmented Urban Navigation","primary_cat":"cs.RO","submitted_at":"2026-06-18T16:40:07+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"A training-free fusion layer enables stale VLM selections to improve a real-time planner's trajectory scoring for urban sidewalk navigation, yielding 30% ADE reduction in challenging scenarios.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.18112","ref_index":27,"ref_count":2,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Qwen-RobotNav Technical Report: A Scalable Navigation Model Designed for an Agentic Navigation System","primary_cat":"cs.RO","submitted_at":"2026-06-16T16:17:44+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Qwen-RobotNav provides a parameterized navigation model trained on 15.6M samples with vision-language co-training that achieves SOTA results on benchmarks and zero-shot transfer to real robots.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.12603","ref_index":30,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"From Imitation to Alignment: Human-Preference Flow Policies for Long-Horizon Sidewalk Navigation","primary_cat":"cs.RO","submitted_at":"2026-06-10T19:01:31+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"FlowPilot combines anchored flow matching for multimodal action pre-training with human-in-the-loop preference learning to improve long-horizon monocular sidewalk navigation, reporting 42% success in simulation and reduced interruptions in real-world tests.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.07244","ref_index":42,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Beyond Waypoints: A Trajectory-Centric Waypointing Paradigm for Vision-Language Navigation","primary_cat":"cs.RO","submitted_at":"2026-06-05T13:11:39+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"The paper introduces a Trajectory Waypoint paradigm with a TSDF-guided diffusion policy and trajectory-enhanced navigator that achieves better performance on VLN-CE benchmarks by ensuring waypoint reachability and planning-execution consistency.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.03175","ref_index":60,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Ask When It Pays: Cost-Aware Open-Ended Interaction for Instance Goal Navigation","primary_cat":"cs.CV","submitted_at":"2026-06-02T05:31:03+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":4.0,"formal_verification":"none","one_line_summary":"Proposes cost-aware question selection for ambiguous object navigation via information-gain analysis on corpora, a cost-penalizing benchmark, and a zero-shot MLLM agent.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.01621","ref_index":26,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Goal2Pixel: Grounding Goals to Pixels for Vision-Language Navigation","primary_cat":"cs.CV","submitted_at":"2026-06-01T03:12:58+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Goal2Pixel grounds VLN-CE goals to image pixels via VLM prediction plus keyframe memory, reaching 54.1% SR on R2R-CE Val-Unseen with 7.75 calls per episode versus 46.62 for action prediction.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.28237","ref_index":29,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"POINav: Benchmarking and Enhancing Final-Meters Arrival in Real-World Vision-Language Navigation","primary_cat":"cs.RO","submitted_at":"2026-05-27T09:50:16+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"POINav-Bench provides the first high-fidelity real-world benchmark for POI-goal VLN using 3DGS reconstructions of 126k m² with 163 POIs, supported by a Brain-Action framework and 70K real signage-entrance dataset.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.27582","ref_index":46,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Uni-LaViRA: Language-Vision-Robot Actions Translation for Unified Embodied Navigation","primary_cat":"cs.RO","submitted_at":"2026-05-26T18:52:04+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"A zero-shot unified agent for VLN-CE, ObjectNav, EQA and Aerial-VLN on wheeled, quadruped, humanoid and UAV platforms that translates language and vision inputs into actions via MLLMs plus TDM and SCB mechanisms, matching trained foundation models on multiple benchmarks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.19420","ref_index":29,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Beyond Waypoints: Dual-Heatmap Grounding for Cross-Embodiment Semantic Navigation","primary_cat":"cs.RO","submitted_at":"2026-05-19T06:12:59+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"A vision-language model outputs dual heatmaps for navigation affordance and facing to ground semantic instructions into executable free space, achieving higher affordance rates than waypoint regression across simulated robot embodiments.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.13328","ref_index":12,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"What Limits Vision-and-Language Navigation ?","primary_cat":"cs.RO","submitted_at":"2026-05-13T10:41:24+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"StereoNav reaches new benchmark highs on R2R-CE and RxR-CE and improves real-robot reliability by supplying persistent target-location priors and stereo-derived geometry that stay stable under lighting changes and blur.","context_count":1,"top_context_role":"baseline","top_context_polarity":"baseline","context_text":"Egocentric RGB AgentNaVid [5] 7B ✗ ✗ ✓ ✗ ✓ 5.5 49.1 37.4 35.9 - - -Uni-NaVid [17] 7B✗ ✗ ✓ ✗ ✓ 5.6 53.3 47.0 42.7 6.2 48.7 40.9NaVILA [4] 8B ✗ ✗ ✓ ✗ ✓ 5.2 62.5 54.0 49.0 6.8 49.3 44.0StreamVLN [6] 7B✗ ✗ ✓ ✗ ✓ 5.0 64.2 56.9 51.9 6.2 52.9 46.0InternVLA-N1 [15] 8B✗ ✗ ✓ ✗ ✓ 4.9 60.6 55.4 52.1 6.4 49.5 41.8NavFoM [16] 7B✗ ✗ ✓ ✗ ✓ 5.0 64.9 56.2 51.2 5.5 57.4 49.4DualVLN [12] 8B✗ ✗ ✓ ✗ ✓ 4.1 70.7 64.3 58.5 4.6 61.4 51.8Efficient-VLN [48] 4B✗ ✗ ✓ ✗ ✓ 4.2 73.7 64.2 55.9 3.9 67.0 54.3JanusVLN [18] 8B✗ ✗ ✓ ✗ ✓ 4.8 65.2 60.5 56.8 6.1 56.2 47.5PROSPECT [13] 9B✗ ✗ ✓ ✓ ✓ 4.9 65.2 58.9 54.0 5.7 54.6 46.2SACA [49] 8B ✗ ✗ ✓ ✗ ✓ 4.2 69.3 64.7 56.9 4.8 62.1 51.7NaVIDA [50] 3B✗ ✗ ✓ ✗ ✓ 4.3 69.5 61.4 54.7 5.2 57.4 49.6DyGeoVLN [14] 9B✗ ✗ ✓ ✓ ✓ 4."},{"citing_arxiv_id":"2605.09441","ref_index":31,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Beyond Isolation: A Unified Benchmark for General-Purpose Navigation","primary_cat":"cs.RO","submitted_at":"2026-05-10T09:34:05+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"OmniNavBench is a unified benchmark for general-purpose navigation featuring composite multi-skill instructions, support for humanoid, quadrupedal and wheeled robots, and 1779 human teleoperated trajectories across 170 environments.","context_count":1,"top_context_role":"background","top_context_polarity":"background","context_text":"efforts demonstrate that current model architectures possess strong potential for cross-task generalization. In parallel, there is a growing trend toward developing generalist algorithms that adapt to different robot morphologies. NavFoM [40] has been deployed across diverse platforms, including quadrupeds, drones, and wheeled robots. InternVLA-N1 [31] showcases zero-shot cross-embodiment generalization, spanning wheeled to humanoid robots, and NaVILA [7] achieves transfer for legged robots by decoupling high-level planning from low- level control. Furthermore, VLN-PE [28] highlights the im- portance of physical embodiment, revealing how sensitive existing algorithms are to robot dynamics and camera con-"},{"citing_arxiv_id":"2604.27620","ref_index":69,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"SpaAct: Spatially-Activated Transition Learning with Curriculum Adaptation for Vision-Language Navigation","primary_cat":"cs.CV","submitted_at":"2026-04-30T09:09:40+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"SpaAct activates spatial awareness in VLMs using action retrospection, future frame prediction, and progressive curriculum learning to reach SOTA on VLN-CE benchmarks.","context_count":1,"top_context_role":"method","top_context_polarity":"use_method","context_text":"low-level action must corresponds to fine-grained control (e.g., moving0 .25𝑚, rotating15 ◦ or stop). The episode terminates when the agent issues the stop command or reaches a maximum step limit 𝑇 . Success is defined by whether the final position𝑝𝑇 is within a distance threshold𝑑 𝑠𝑢𝑐𝑐𝑒𝑠𝑠 from the ground-truth target𝑝 ∗. Backbone.We follow JanusVLN [ 69] and adopt a VLM-based nav- igation architecture as our backbone. Given the language instruc- tion I, current egocentric observation 𝑜𝑡 , and navigation history H𝑡 , the model autoregressively predicts the next low-level action: 𝑝𝜃 (𝑎𝑡 | I,H 𝑡, 𝑜𝑡 )=LLM E(I),E(H 𝑡 , 𝑜𝑡 )\u0001,(1) whereEdenotes the embedded tokens of the instruction, history, and current observation."},{"citing_arxiv_id":"2604.17473","ref_index":115,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Dual-Anchoring: Addressing State Drift in Vision-Language Navigation","primary_cat":"cs.CV","submitted_at":"2026-04-19T15:03:38+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Privatar partitions VR avatar reconstruction via frequency-domain decomposition, keeping sensitive components local and offloading the rest with distribution-aware minimal perturbation noise, achieving 2.37x throughput with provable privacy.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2603.26788","ref_index":36,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"ReMemNav: A Rethinking and Memory-Augmented Framework for Zero-Shot Object Navigation","primary_cat":"cs.RO","submitted_at":"2026-03-25T09:07:32+00:00","verdict":"CONDITIONAL","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"ReMemNav improves zero-shot object navigation success and efficiency by integrating episodic memory and rethinking with VLMs, achieving SR/SPL gains of 1.7%/7.0% on HM3D v0.1, 18.2%/11.1% on HM3D v0.2, and 8.7%/7.9% on MP3D.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2603.07080","ref_index":29,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"VLN-Cache: Enabling Token Caching for VLN Models with Visual/Semantic Dynamics Awareness","primary_cat":"cs.RO","submitted_at":"2026-03-07T07:30:35+00:00","verdict":"CONDITIONAL","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"VLN-Cache delivers up to 1.52x faster inference in VLN models by using view-aligned remapping for geometric consistency and a task-relevance saliency filter to manage semantic changes during navigation.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null}],"limit":50,"offset":0}