{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:372IEFS5FWMNYF5PJTFL2UF5AB","short_pith_number":"pith:372IEFS5","schema_version":"1.0","canonical_sha256":"dff482165d2d98dc17af4ccabd50bd0056dd135a20d1e66f73551bd6473458fc","source":{"kind":"arxiv","id":"2406.04882","version":1},"attestation_state":"computed","paper":{"title":"InstructNav: Zero-shot System for Generic Instruction Navigation in Unexplored Environment","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CV"],"primary_cat":"cs.RO","authors_text":"Guanqi Zhan, Hao Dong, Hongcheng Wang, Wenzhe Cai, Yuxing Long","submitted_at":"2024-06-07T12:26:34Z","abstract_excerpt":"Enabling robots to navigate following diverse language instructions in unexplored environments is an attractive goal for human-robot interaction. However, this goal is challenging because different navigation tasks require different strategies. The scarcity of instruction navigation data hinders training an instruction navigation model with varied strategies. Therefore, previous methods are all constrained to one specific type of navigation instruction. In this work, we propose InstructNav, a generic instruction navigation system. InstructNav makes the first endeavor to handle various instruct"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.04882","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2024-06-07T12:26:34Z","cross_cats_sorted":["cs.AI","cs.CL","cs.CV"],"title_canon_sha256":"66a1dc861c1732e9663685eb2b3b99224071972afc07461e844e0db2a6b1d656","abstract_canon_sha256":"7ab339feb18357e2485d2cd8a11591d074a37b2ffee0b36d4ab483e22f1e7a23"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:28:50.415980Z","signature_b64":"Ev17ZU4IMsLlkOh+9bYwffzI/28vG64AlbrTWvLxv/Bj9RZQ7GdWxa8ESLjLqJ+E5bd9v9/5SP2bTvYCDy53AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dff482165d2d98dc17af4ccabd50bd0056dd135a20d1e66f73551bd6473458fc","last_reissued_at":"2026-07-05T08:28:50.415561Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:28:50.415561Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"InstructNav: Zero-shot System for Generic Instruction Navigation in Unexplored Environment","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CV"],"primary_cat":"cs.RO","authors_text":"Guanqi Zhan, Hao Dong, Hongcheng Wang, Wenzhe Cai, Yuxing Long","submitted_at":"2024-06-07T12:26:34Z","abstract_excerpt":"Enabling robots to navigate following diverse language instructions in unexplored environments is an attractive goal for human-robot interaction. However, this goal is challenging because different navigation tasks require different strategies. The scarcity of instruction navigation data hinders training an instruction navigation model with varied strategies. Therefore, previous methods are all constrained to one specific type of navigation instruction. In this work, we propose InstructNav, a generic instruction navigation system. InstructNav makes the first endeavor to handle various instruct"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.04882","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.04882/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.04882","created_at":"2026-07-05T08:28:50.415626+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.04882v1","created_at":"2026-07-05T08:28:50.415626+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.04882","created_at":"2026-07-05T08:28:50.415626+00:00"},{"alias_kind":"pith_short_12","alias_value":"372IEFS5FWMN","created_at":"2026-07-05T08:28:50.415626+00:00"},{"alias_kind":"pith_short_16","alias_value":"372IEFS5FWMNYF5P","created_at":"2026-07-05T08:28:50.415626+00:00"},{"alias_kind":"pith_short_8","alias_value":"372IEFS5","created_at":"2026-07-05T08:28:50.415626+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":31,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25119","citing_title":"SurveilNav: Collaborative Object Goal Navigation with Robot and Surveillance System","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22424","citing_title":"FlowDec: Temporal Conditional Flow Decorruptor for Robust Continuous Vision-Language Navigation","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02417","citing_title":"LIME: Learning Intent-aware Camera Motion from Egocentric Video","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10927","citing_title":"AllDayNav: Lifelong Navigation via Real-World Reinforcement Learning","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07244","citing_title":"Beyond Waypoints: A Trajectory-Centric Waypointing Paradigm for Vision-Language Navigation","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03175","citing_title":"Ask When It Pays: Cost-Aware Open-Ended Interaction for Instance Goal Navigation","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01621","citing_title":"Goal2Pixel: Grounding Goals to Pixels for Vision-Language Navigation","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17249","citing_title":"SEDualVLN: A Spatially-Enhanced Dual-System for Vision-Language Navigation","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30367","citing_title":"FutureNav: Unified World-Action Modeling for Vision-and-Language Navigation","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27807","citing_title":"SpikeVLA: Vision-Language-Action Models with Spiking Neural Networks","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28237","citing_title":"POINav: Benchmarking and Enhancing Final-Meters Arrival in Real-World Vision-Language Navigation","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2502.13451","citing_title":"MapNav: A Novel Memory Representation via Annotated Semantic Maps for Vision-and-Language Navigation","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22036","citing_title":"GA-VLN: Geometry-Aware BEV Representation for Efficient Vision-Language Navigation","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2603.16947","citing_title":"LightZeroNav: Zero-Shot Vision Language Navigation in Continuous Environments Based on Lightweight VLMs","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17249","citing_title":"SEDualVLN: A Spatially-Enhanced Dual-System for Vision-Language Navigation","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19634","citing_title":"P2DNav: Panorama-to-Downview Reasoning for Zero-shot Vision-and-Language Navigation","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16979","citing_title":"NORM-Nav: Zero-Shot Mobile Robot Navigation with Natural Language Behavioral Constraints","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06223","citing_title":"ProCompNav: Proactive Instance Navigation with Comparative Judgment for Ambiguous User Queries","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2511.17207","citing_title":"SING3R-SLAM: Submap-based Indoor Monocular Gaussian SLAM with 3D Reconstruction Priors","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2412.06224","citing_title":"Uni-NaVid: A Video-based Vision-Language-Action Model for Unifying Embodied Navigation Tasks","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2512.21714","citing_title":"AstraNav-World: World Model for Foresight Control and Consistency","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2603.20530","citing_title":"Memory Over Maps: 3D Object Localization Without Reconstruction","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2603.26788","citing_title":"ReMemNav: A Rethinking and Memory-Augmented Framework for Zero-Shot Object Navigation","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02829","citing_title":"STRNet: Visual Navigation with Spatio-Temporal Representation through Dynamic Graph Aggregation","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27620","citing_title":"SpaAct: Spatially-Activated Transition Learning with Curriculum Adaptation for Vision-Language Navigation","ref_index":46,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/372IEFS5FWMNYF5PJTFL2UF5AB","json":"https://pith.science/pith/372IEFS5FWMNYF5PJTFL2UF5AB.json","graph_json":"https://pith.science/api/pith-number/372IEFS5FWMNYF5PJTFL2UF5AB/graph.json","events_json":"https://pith.science/api/pith-number/372IEFS5FWMNYF5PJTFL2UF5AB/events.json","paper":"https://pith.science/paper/372IEFS5"},"agent_actions":{"view_html":"https://pith.science/pith/372IEFS5FWMNYF5PJTFL2UF5AB","download_json":"https://pith.science/pith/372IEFS5FWMNYF5PJTFL2UF5AB.json","view_paper":"https://pith.science/paper/372IEFS5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.04882&json=true","fetch_graph":"https://pith.science/api/pith-number/372IEFS5FWMNYF5PJTFL2UF5AB/graph.json","fetch_events":"https://pith.science/api/pith-number/372IEFS5FWMNYF5PJTFL2UF5AB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/372IEFS5FWMNYF5PJTFL2UF5AB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/372IEFS5FWMNYF5PJTFL2UF5AB/action/storage_attestation","attest_author":"https://pith.science/pith/372IEFS5FWMNYF5PJTFL2UF5AB/action/author_attestation","sign_citation":"https://pith.science/pith/372IEFS5FWMNYF5PJTFL2UF5AB/action/citation_signature","submit_replication":"https://pith.science/pith/372IEFS5FWMNYF5PJTFL2UF5AB/action/replication_record"}},"created_at":"2026-07-05T08:28:50.415626+00:00","updated_at":"2026-07-05T08:28:50.415626+00:00"}