{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:YMPD6V5HNDAHRSMKO5PT7ON46L","short_pith_number":"pith:YMPD6V5H","schema_version":"1.0","canonical_sha256":"c31e3f57a768c078c98a775f3fb9bcf2dd1d1fc2bdef0cb781086cff0732963b","source":{"kind":"arxiv","id":"2310.10822","version":1},"attestation_state":"computed","paper":{"title":"Vision and Language Navigation in the Real World via Online Visual Language Mapping","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.CV","cs.SY","eess.SY"],"primary_cat":"cs.RO","authors_text":"Chengguang Xu, Christopher Amato, Hieu T. Nguyen, Lawson L.S. Wong","submitted_at":"2023-10-16T20:44:09Z","abstract_excerpt":"Navigating in unseen environments is crucial for mobile robots. Enhancing them with the ability to follow instructions in natural language will further improve navigation efficiency in unseen cases. However, state-of-the-art (SOTA) vision-and-language navigation (VLN) methods are mainly evaluated in simulation, neglecting the complex and noisy real world. Directly transferring SOTA navigation policies trained in simulation to the real world is challenging due to the visual domain gap and the absence of prior knowledge about unseen environments. In this work, we propose a novel navigation frame"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.10822","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.RO","submitted_at":"2023-10-16T20:44:09Z","cross_cats_sorted":["cs.CV","cs.SY","eess.SY"],"title_canon_sha256":"d1b84eac1a9f72d0845a1de6f4e94410409cda8fa74dcb406ef0dea5f2b29111","abstract_canon_sha256":"d3a92cb1c6e8c49bdfd52488d34852e230051ca297241c8fba80e0522064f6a3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:01:28.349881Z","signature_b64":"PomI1EYEXx8pQW/Xw+U7eJafVdraFpjv363071UUml5JOSeB6OsFnHuQ3UtqDDPd2Hfc0fxQzZuge3xIz1R4Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c31e3f57a768c078c98a775f3fb9bcf2dd1d1fc2bdef0cb781086cff0732963b","last_reissued_at":"2026-07-05T07:01:28.349393Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:01:28.349393Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Vision and Language Navigation in the Real World via Online Visual Language Mapping","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.CV","cs.SY","eess.SY"],"primary_cat":"cs.RO","authors_text":"Chengguang Xu, Christopher Amato, Hieu T. Nguyen, Lawson L.S. Wong","submitted_at":"2023-10-16T20:44:09Z","abstract_excerpt":"Navigating in unseen environments is crucial for mobile robots. Enhancing them with the ability to follow instructions in natural language will further improve navigation efficiency in unseen cases. However, state-of-the-art (SOTA) vision-and-language navigation (VLN) methods are mainly evaluated in simulation, neglecting the complex and noisy real world. Directly transferring SOTA navigation policies trained in simulation to the real world is challenging due to the visual domain gap and the absence of prior knowledge about unseen environments. In this work, we propose a novel navigation frame"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.10822","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.10822/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.10822","created_at":"2026-07-05T07:01:28.349451+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.10822v1","created_at":"2026-07-05T07:01:28.349451+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.10822","created_at":"2026-07-05T07:01:28.349451+00:00"},{"alias_kind":"pith_short_12","alias_value":"YMPD6V5HNDAH","created_at":"2026-07-05T07:01:28.349451+00:00"},{"alias_kind":"pith_short_16","alias_value":"YMPD6V5HNDAHRSMK","created_at":"2026-07-05T07:01:28.349451+00:00"},{"alias_kind":"pith_short_8","alias_value":"YMPD6V5H","created_at":"2026-07-05T07:01:28.349451+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20641","citing_title":"MAGNIFIED: RL Fine-tuning of Multimodal Large Language Models for Motion Planning","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01565","citing_title":"Hierarchical Semantic-Augmented Navigation: Optimal Transport and Graph-Driven Reasoning for Vision-Language Navigation","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2402.15852","citing_title":"NaVid: Video-based VLM Plans the Next Step for Vision-and-Language Navigation","ref_index":108,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YMPD6V5HNDAHRSMKO5PT7ON46L","json":"https://pith.science/pith/YMPD6V5HNDAHRSMKO5PT7ON46L.json","graph_json":"https://pith.science/api/pith-number/YMPD6V5HNDAHRSMKO5PT7ON46L/graph.json","events_json":"https://pith.science/api/pith-number/YMPD6V5HNDAHRSMKO5PT7ON46L/events.json","paper":"https://pith.science/paper/YMPD6V5H"},"agent_actions":{"view_html":"https://pith.science/pith/YMPD6V5HNDAHRSMKO5PT7ON46L","download_json":"https://pith.science/pith/YMPD6V5HNDAHRSMKO5PT7ON46L.json","view_paper":"https://pith.science/paper/YMPD6V5H","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.10822&json=true","fetch_graph":"https://pith.science/api/pith-number/YMPD6V5HNDAHRSMKO5PT7ON46L/graph.json","fetch_events":"https://pith.science/api/pith-number/YMPD6V5HNDAHRSMKO5PT7ON46L/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YMPD6V5HNDAHRSMKO5PT7ON46L/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YMPD6V5HNDAHRSMKO5PT7ON46L/action/storage_attestation","attest_author":"https://pith.science/pith/YMPD6V5HNDAHRSMKO5PT7ON46L/action/author_attestation","sign_citation":"https://pith.science/pith/YMPD6V5HNDAHRSMKO5PT7ON46L/action/citation_signature","submit_replication":"https://pith.science/pith/YMPD6V5HNDAHRSMKO5PT7ON46L/action/replication_record"}},"created_at":"2026-07-05T07:01:28.349451+00:00","updated_at":"2026-07-05T07:01:28.349451+00:00"}