{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:HVLO7EQFTBE5B4KGCZ64FGGI3B","short_pith_number":"pith:HVLO7EQF","schema_version":"1.0","canonical_sha256":"3d56ef92059849d0f146167dc298c8d85b12528877f5575611dddc72d003fd67","source":{"kind":"arxiv","id":"2208.11781","version":1},"attestation_state":"computed","paper":{"title":"Learning from Unlabeled 3D Environments for Vision-and-Language Navigation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Cordelia Schmid, Ivan Laptev, Makarand Tapaswi, Pierre-Louis Guhur, Shizhe Chen","submitted_at":"2022-08-24T21:50:20Z","abstract_excerpt":"In vision-and-language navigation (VLN), an embodied agent is required to navigate in realistic 3D environments following natural language instructions. One major bottleneck for existing VLN approaches is the lack of sufficient training data, resulting in unsatisfactory generalization to unseen environments. While VLN data is typically collected manually, such an approach is expensive and prevents scalability. In this work, we address the data scarcity issue by proposing to automatically create a large-scale VLN dataset from 900 unlabeled 3D buildings from HM3D. We generate a navigation graph "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2208.11781","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2022-08-24T21:50:20Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"4be8d16471b7531c51de7f78900e48300cb49b79d8b65e640eca3ce247fbac31","abstract_canon_sha256":"45516939bda176eaf184a49ee334e9f55379470fb6d32f0fe001c113aee1f8ee"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:51:25.960408Z","signature_b64":"+S4PomCZiLKhy7FS9jXSq+9FsKxZ5vRXF+DSt8boLo0tUq9pFtVbUg3rqKFT8YBteoVuIwkNQeJFGNx6aBx4Cw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3d56ef92059849d0f146167dc298c8d85b12528877f5575611dddc72d003fd67","last_reissued_at":"2026-07-05T04:51:25.959922Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:51:25.959922Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning from Unlabeled 3D Environments for Vision-and-Language Navigation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Cordelia Schmid, Ivan Laptev, Makarand Tapaswi, Pierre-Louis Guhur, Shizhe Chen","submitted_at":"2022-08-24T21:50:20Z","abstract_excerpt":"In vision-and-language navigation (VLN), an embodied agent is required to navigate in realistic 3D environments following natural language instructions. One major bottleneck for existing VLN approaches is the lack of sufficient training data, resulting in unsatisfactory generalization to unseen environments. While VLN data is typically collected manually, such an approach is expensive and prevents scalability. In this work, we address the data scarcity issue by proposing to automatically create a large-scale VLN dataset from 900 unlabeled 3D buildings from HM3D. We generate a navigation graph "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2208.11781","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2208.11781/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2208.11781","created_at":"2026-07-05T04:51:25.959974+00:00"},{"alias_kind":"arxiv_version","alias_value":"2208.11781v1","created_at":"2026-07-05T04:51:25.959974+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2208.11781","created_at":"2026-07-05T04:51:25.959974+00:00"},{"alias_kind":"pith_short_12","alias_value":"HVLO7EQFTBE5","created_at":"2026-07-05T04:51:25.959974+00:00"},{"alias_kind":"pith_short_16","alias_value":"HVLO7EQFTBE5B4KG","created_at":"2026-07-05T04:51:25.959974+00:00"},{"alias_kind":"pith_short_8","alias_value":"HVLO7EQF","created_at":"2026-07-05T04:51:25.959974+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.16654","citing_title":"MSNav: Zero-Shot Vision-and-Language Navigation with Dynamic Memory and LLM Spatial Reasoning","ref_index":10,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HVLO7EQFTBE5B4KGCZ64FGGI3B","json":"https://pith.science/pith/HVLO7EQFTBE5B4KGCZ64FGGI3B.json","graph_json":"https://pith.science/api/pith-number/HVLO7EQFTBE5B4KGCZ64FGGI3B/graph.json","events_json":"https://pith.science/api/pith-number/HVLO7EQFTBE5B4KGCZ64FGGI3B/events.json","paper":"https://pith.science/paper/HVLO7EQF"},"agent_actions":{"view_html":"https://pith.science/pith/HVLO7EQFTBE5B4KGCZ64FGGI3B","download_json":"https://pith.science/pith/HVLO7EQFTBE5B4KGCZ64FGGI3B.json","view_paper":"https://pith.science/paper/HVLO7EQF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2208.11781&json=true","fetch_graph":"https://pith.science/api/pith-number/HVLO7EQFTBE5B4KGCZ64FGGI3B/graph.json","fetch_events":"https://pith.science/api/pith-number/HVLO7EQFTBE5B4KGCZ64FGGI3B/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HVLO7EQFTBE5B4KGCZ64FGGI3B/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HVLO7EQFTBE5B4KGCZ64FGGI3B/action/storage_attestation","attest_author":"https://pith.science/pith/HVLO7EQFTBE5B4KGCZ64FGGI3B/action/author_attestation","sign_citation":"https://pith.science/pith/HVLO7EQFTBE5B4KGCZ64FGGI3B/action/citation_signature","submit_replication":"https://pith.science/pith/HVLO7EQFTBE5B4KGCZ64FGGI3B/action/replication_record"}},"created_at":"2026-07-05T04:51:25.959974+00:00","updated_at":"2026-07-05T04:51:25.959974+00:00"}