{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:SQZNZFIOIJB5RUZXRTZETGGN3R","short_pith_number":"pith:SQZNZFIO","schema_version":"1.0","canonical_sha256":"9432dc950e4243d8d3378cf24998cddc4fe9f8d8a3c91c7c6c8aeaf00fd82750","source":{"kind":"arxiv","id":"2310.07889","version":2},"attestation_state":"computed","paper":{"title":"LangNav: Language as a Perceptual Representation for Navigation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.RO"],"primary_cat":"cs.CV","authors_text":"Aude Oliva, Bowen Pan, Phillip Isola, Rameswar Panda, Rogerio Feris, SouYoung Jin, Yoon Kim","submitted_at":"2023-10-11T20:52:30Z","abstract_excerpt":"We explore the use of language as a perceptual representation for vision-and-language navigation (VLN), with a focus on low-data settings. Our approach uses off-the-shelf vision systems for image captioning and object detection to convert an agent's egocentric panoramic view at each time step into natural language descriptions. We then finetune a pretrained language model to select an action, based on the current view and the trajectory history, that would best fulfill the navigation instructions. In contrast to the standard setup which adapts a pretrained language model to work directly with "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.07889","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-10-11T20:52:30Z","cross_cats_sorted":["cs.AI","cs.CL","cs.RO"],"title_canon_sha256":"8266ce462931daf2281b3abee4a56d210b1939a1b23617732e8170eed3e52b67","abstract_canon_sha256":"c2b5dac04c25e24d025a8930089da8cd41efda8d41e4116f6b9aca16057a8bcc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:02:22.020830Z","signature_b64":"cOiNSN4WWhn3vnP3uulHHDpPlEDNftX4L6kxVz5eUfmPVuSqYdrrK3CngVUsveK4IaMmie7/AGSSPBS0cj/RDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9432dc950e4243d8d3378cf24998cddc4fe9f8d8a3c91c7c6c8aeaf00fd82750","last_reissued_at":"2026-07-05T08:02:22.020323Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:02:22.020323Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LangNav: Language as a Perceptual Representation for Navigation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.RO"],"primary_cat":"cs.CV","authors_text":"Aude Oliva, Bowen Pan, Phillip Isola, Rameswar Panda, Rogerio Feris, SouYoung Jin, Yoon Kim","submitted_at":"2023-10-11T20:52:30Z","abstract_excerpt":"We explore the use of language as a perceptual representation for vision-and-language navigation (VLN), with a focus on low-data settings. Our approach uses off-the-shelf vision systems for image captioning and object detection to convert an agent's egocentric panoramic view at each time step into natural language descriptions. We then finetune a pretrained language model to select an action, based on the current view and the trajectory history, that would best fulfill the navigation instructions. In contrast to the standard setup which adapts a pretrained language model to work directly with "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.07889","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.07889/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.07889","created_at":"2026-07-05T08:02:22.020385+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.07889v2","created_at":"2026-07-05T08:02:22.020385+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.07889","created_at":"2026-07-05T08:02:22.020385+00:00"},{"alias_kind":"pith_short_12","alias_value":"SQZNZFIOIJB5","created_at":"2026-07-05T08:02:22.020385+00:00"},{"alias_kind":"pith_short_16","alias_value":"SQZNZFIOIJB5RUZX","created_at":"2026-07-05T08:02:22.020385+00:00"},{"alias_kind":"pith_short_8","alias_value":"SQZNZFIO","created_at":"2026-07-05T08:02:22.020385+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22424","citing_title":"FlowDec: Temporal Conditional Flow Decorruptor for Robust Continuous Vision-Language Navigation","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03175","citing_title":"Ask When It Pays: Cost-Aware Open-Ended Interaction for Instance Goal Navigation","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2402.15852","citing_title":"NaVid: Video-based VLM Plans the Next Step for Vision-and-Language Navigation","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2603.02972","citing_title":"TagaVLM: Topology-Aware Global Action Reasoning for Vision-Language Navigation","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10533","citing_title":"VLN-NF: Feasibility-Aware Vision-and-Language Navigation with False-Premise Instructions","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SQZNZFIOIJB5RUZXRTZETGGN3R","json":"https://pith.science/pith/SQZNZFIOIJB5RUZXRTZETGGN3R.json","graph_json":"https://pith.science/api/pith-number/SQZNZFIOIJB5RUZXRTZETGGN3R/graph.json","events_json":"https://pith.science/api/pith-number/SQZNZFIOIJB5RUZXRTZETGGN3R/events.json","paper":"https://pith.science/paper/SQZNZFIO"},"agent_actions":{"view_html":"https://pith.science/pith/SQZNZFIOIJB5RUZXRTZETGGN3R","download_json":"https://pith.science/pith/SQZNZFIOIJB5RUZXRTZETGGN3R.json","view_paper":"https://pith.science/paper/SQZNZFIO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.07889&json=true","fetch_graph":"https://pith.science/api/pith-number/SQZNZFIOIJB5RUZXRTZETGGN3R/graph.json","fetch_events":"https://pith.science/api/pith-number/SQZNZFIOIJB5RUZXRTZETGGN3R/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SQZNZFIOIJB5RUZXRTZETGGN3R/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SQZNZFIOIJB5RUZXRTZETGGN3R/action/storage_attestation","attest_author":"https://pith.science/pith/SQZNZFIOIJB5RUZXRTZETGGN3R/action/author_attestation","sign_citation":"https://pith.science/pith/SQZNZFIOIJB5RUZXRTZETGGN3R/action/citation_signature","submit_replication":"https://pith.science/pith/SQZNZFIOIJB5RUZXRTZETGGN3R/action/replication_record"}},"created_at":"2026-07-05T08:02:22.020385+00:00","updated_at":"2026-07-05T08:02:22.020385+00:00"}