{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:P2BJX576T6UOZQVCWY2LO5SLVK","short_pith_number":"pith:P2BJX576","schema_version":"1.0","canonical_sha256":"7e829bf7fe9fa8ecc2a2b634b7764baaa5580d85d374a6a363090feb64aa5a2f","source":{"kind":"arxiv","id":"2108.11544","version":3},"attestation_state":"computed","paper":{"title":"Vision-Language Navigation: A Survey and Taxonomy","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.MM"],"primary_cat":"cs.CV","authors_text":"Tao Chang, Wansen Wu, Xinmeng Li","submitted_at":"2021-08-26T01:51:18Z","abstract_excerpt":"Vision-Language Navigation (VLN) tasks require an agent to follow human language instructions to navigate in previously unseen environments. This challenging field involving problems in natural language processing, computer vision, robotics, etc., has spawn many excellent works focusing on various VLN tasks. This paper provides a comprehensive survey and an insightful taxonomy of these tasks based on the different characteristics of language instructions in these tasks. Depending on whether the navigation instructions are given for once or multiple times, this paper divides the tasks into two "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2108.11544","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2021-08-26T01:51:18Z","cross_cats_sorted":["cs.MM"],"title_canon_sha256":"4ba53e06186c68f9f7b961d20a5913fee3eade3aeadbc09c70a71ef19217caea","abstract_canon_sha256":"e34163aa7d686a5cd66422153f0f19554be2f42c88884fd33c571a1ed69d027d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:10:51.006329Z","signature_b64":"XOrqwOTY74I5MlXWFWkH2W3SFK0YzmuItXAKGn9fSqxiX0e86fzt0LaS6mzJ07MME3ygh/WTLbpl9TWS6l+PDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7e829bf7fe9fa8ecc2a2b634b7764baaa5580d85d374a6a363090feb64aa5a2f","last_reissued_at":"2026-07-05T04:10:51.005917Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:10:51.005917Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Vision-Language Navigation: A Survey and Taxonomy","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.MM"],"primary_cat":"cs.CV","authors_text":"Tao Chang, Wansen Wu, Xinmeng Li","submitted_at":"2021-08-26T01:51:18Z","abstract_excerpt":"Vision-Language Navigation (VLN) tasks require an agent to follow human language instructions to navigate in previously unseen environments. This challenging field involving problems in natural language processing, computer vision, robotics, etc., has spawn many excellent works focusing on various VLN tasks. This paper provides a comprehensive survey and an insightful taxonomy of these tasks based on the different characteristics of language instructions in these tasks. Depending on whether the navigation instructions are given for once or multiple times, this paper divides the tasks into two "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2108.11544","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2108.11544/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2108.11544","created_at":"2026-07-05T04:10:51.005979+00:00"},{"alias_kind":"arxiv_version","alias_value":"2108.11544v3","created_at":"2026-07-05T04:10:51.005979+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2108.11544","created_at":"2026-07-05T04:10:51.005979+00:00"},{"alias_kind":"pith_short_12","alias_value":"P2BJX576T6UO","created_at":"2026-07-05T04:10:51.005979+00:00"},{"alias_kind":"pith_short_16","alias_value":"P2BJX576T6UOZQVC","created_at":"2026-07-05T04:10:51.005979+00:00"},{"alias_kind":"pith_short_8","alias_value":"P2BJX576","created_at":"2026-07-05T04:10:51.005979+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.19064","citing_title":"The Essence of Balance for Self-Improving Agents in Vision-and-Language Navigation","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/P2BJX576T6UOZQVCWY2LO5SLVK","json":"https://pith.science/pith/P2BJX576T6UOZQVCWY2LO5SLVK.json","graph_json":"https://pith.science/api/pith-number/P2BJX576T6UOZQVCWY2LO5SLVK/graph.json","events_json":"https://pith.science/api/pith-number/P2BJX576T6UOZQVCWY2LO5SLVK/events.json","paper":"https://pith.science/paper/P2BJX576"},"agent_actions":{"view_html":"https://pith.science/pith/P2BJX576T6UOZQVCWY2LO5SLVK","download_json":"https://pith.science/pith/P2BJX576T6UOZQVCWY2LO5SLVK.json","view_paper":"https://pith.science/paper/P2BJX576","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2108.11544&json=true","fetch_graph":"https://pith.science/api/pith-number/P2BJX576T6UOZQVCWY2LO5SLVK/graph.json","fetch_events":"https://pith.science/api/pith-number/P2BJX576T6UOZQVCWY2LO5SLVK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/P2BJX576T6UOZQVCWY2LO5SLVK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/P2BJX576T6UOZQVCWY2LO5SLVK/action/storage_attestation","attest_author":"https://pith.science/pith/P2BJX576T6UOZQVCWY2LO5SLVK/action/author_attestation","sign_citation":"https://pith.science/pith/P2BJX576T6UOZQVCWY2LO5SLVK/action/citation_signature","submit_replication":"https://pith.science/pith/P2BJX576T6UOZQVCWY2LO5SLVK/action/replication_record"}},"created_at":"2026-07-05T04:10:51.005979+00:00","updated_at":"2026-07-05T04:10:51.005979+00:00"}