{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:OZHTOMNBSPKYUJHT62E7GIPEC5","short_pith_number":"pith:OZHTOMNB","schema_version":"1.0","canonical_sha256":"764f3731a193d58a24f3f689f321e417587f26bd4c51eeabf42c9d7b4cd72dbb","source":{"kind":"arxiv","id":"2304.03047","version":3},"attestation_state":"computed","paper":{"title":"ETPNav: Evolving Topological Planning for Vision-Language Navigation in Continuous Environments","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.RO"],"primary_cat":"cs.CV","authors_text":"Dong An, Hanqing Wang, Keji He, Liang Wang, Wenguan Wang, Yan Huang, Zun Wang","submitted_at":"2023-04-06T13:07:17Z","abstract_excerpt":"Vision-language navigation is a task that requires an agent to follow instructions to navigate in environments. It becomes increasingly crucial in the field of embodied AI, with potential applications in autonomous navigation, search and rescue, and human-robot interaction. In this paper, we propose to address a more practical yet challenging counterpart setting - vision-language navigation in continuous environments (VLN-CE). To develop a robust VLN-CE agent, we propose a new navigation framework, ETPNav, which focuses on two critical skills: 1) the capability to abstract environments and gen"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2304.03047","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-04-06T13:07:17Z","cross_cats_sorted":["cs.CL","cs.RO"],"title_canon_sha256":"a495c373c0fa09460acaddd8bdc267c5f3b4f64b264281499ff8968120a70e7b","abstract_canon_sha256":"45a8959ca46cea4540d5871bb1726fb57339e0651fb5b20f3c1e155c99e833c9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:35:59.923828Z","signature_b64":"0+uEOkdU5s1SzXAHAyTMm3EUoNpIJnqkC6lejmrV5P/TobPLTjrTRaUcJq4TFdPL2o+1Ftd+E+Aj6QHOAf3YDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"764f3731a193d58a24f3f689f321e417587f26bd4c51eeabf42c9d7b4cd72dbb","last_reissued_at":"2026-07-05T07:35:59.923355Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:35:59.923355Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ETPNav: Evolving Topological Planning for Vision-Language Navigation in Continuous Environments","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.RO"],"primary_cat":"cs.CV","authors_text":"Dong An, Hanqing Wang, Keji He, Liang Wang, Wenguan Wang, Yan Huang, Zun Wang","submitted_at":"2023-04-06T13:07:17Z","abstract_excerpt":"Vision-language navigation is a task that requires an agent to follow instructions to navigate in environments. It becomes increasingly crucial in the field of embodied AI, with potential applications in autonomous navigation, search and rescue, and human-robot interaction. In this paper, we propose to address a more practical yet challenging counterpart setting - vision-language navigation in continuous environments (VLN-CE). To develop a robust VLN-CE agent, we propose a new navigation framework, ETPNav, which focuses on two critical skills: 1) the capability to abstract environments and gen"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2304.03047","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2304.03047/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2304.03047","created_at":"2026-07-05T07:35:59.923416+00:00"},{"alias_kind":"arxiv_version","alias_value":"2304.03047v3","created_at":"2026-07-05T07:35:59.923416+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2304.03047","created_at":"2026-07-05T07:35:59.923416+00:00"},{"alias_kind":"pith_short_12","alias_value":"OZHTOMNBSPKY","created_at":"2026-07-05T07:35:59.923416+00:00"},{"alias_kind":"pith_short_16","alias_value":"OZHTOMNBSPKYUJHT","created_at":"2026-07-05T07:35:59.923416+00:00"},{"alias_kind":"pith_short_8","alias_value":"OZHTOMNB","created_at":"2026-07-05T07:35:59.923416+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21398","citing_title":"BIT-Nav: Brain-Inspired Trajectory Memory for Embodied Navigation","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03175","citing_title":"Ask When It Pays: Cost-Aware Open-Ended Interaction for Instance Goal Navigation","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2402.15852","citing_title":"NaVid: Video-based VLM Plans the Next Step for Vision-and-Language Navigation","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2412.06224","citing_title":"Uni-NaVid: A Video-based Vision-Language-Action Model for Unifying Embodied Navigation Tasks","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OZHTOMNBSPKYUJHT62E7GIPEC5","json":"https://pith.science/pith/OZHTOMNBSPKYUJHT62E7GIPEC5.json","graph_json":"https://pith.science/api/pith-number/OZHTOMNBSPKYUJHT62E7GIPEC5/graph.json","events_json":"https://pith.science/api/pith-number/OZHTOMNBSPKYUJHT62E7GIPEC5/events.json","paper":"https://pith.science/paper/OZHTOMNB"},"agent_actions":{"view_html":"https://pith.science/pith/OZHTOMNBSPKYUJHT62E7GIPEC5","download_json":"https://pith.science/pith/OZHTOMNBSPKYUJHT62E7GIPEC5.json","view_paper":"https://pith.science/paper/OZHTOMNB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2304.03047&json=true","fetch_graph":"https://pith.science/api/pith-number/OZHTOMNBSPKYUJHT62E7GIPEC5/graph.json","fetch_events":"https://pith.science/api/pith-number/OZHTOMNBSPKYUJHT62E7GIPEC5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OZHTOMNBSPKYUJHT62E7GIPEC5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OZHTOMNBSPKYUJHT62E7GIPEC5/action/storage_attestation","attest_author":"https://pith.science/pith/OZHTOMNBSPKYUJHT62E7GIPEC5/action/author_attestation","sign_citation":"https://pith.science/pith/OZHTOMNBSPKYUJHT62E7GIPEC5/action/citation_signature","submit_replication":"https://pith.science/pith/OZHTOMNBSPKYUJHT62E7GIPEC5/action/replication_record"}},"created_at":"2026-07-05T07:35:59.923416+00:00","updated_at":"2026-07-05T07:35:59.923416+00:00"}