{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:K6GBQ4DEOF7IRAGWSPHPEYKJC5","short_pith_number":"pith:K6GBQ4DE","schema_version":"1.0","canonical_sha256":"578c187064717e8880d693cef26149174da084617cdd67bee8aa708ff6bc0e17","source":{"kind":"arxiv","id":"2003.00857","version":3},"attestation_state":"computed","paper":{"title":"Multi-View Learning for Vision-and-Language Navigation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","cs.LG"],"primary_cat":"cs.CL","authors_text":"Chunyuan Li, Jianfeng Gao, Noah A. Smith, Qiaolin Xia, Xiujun Li, Yejin Choi, Yonatan Bisk, Zhifang Sui","submitted_at":"2020-03-02T13:07:46Z","abstract_excerpt":"Learning to navigate in a visual environment following natural language instructions is a challenging task because natural language instructions are highly variable, ambiguous, and under-specified. In this paper, we present a novel training paradigm, Learn from EveryOne (LEO), which leverages multiple instructions (as different views) for the same trajectory to resolve language ambiguity and improve generalization. By sharing parameters across instructions, our approach learns more effectively from limited training data and generalizes better in unseen environments. On the recent Room-to-Room "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2003.00857","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2020-03-02T13:07:46Z","cross_cats_sorted":["cs.CV","cs.LG"],"title_canon_sha256":"3fff72d57b397f6d2bd42b277064422914d491e4a1e3a5f40a441de105571494","abstract_canon_sha256":"46e2373fa11d28d73a72ba0b3e3f3bf32a775b9a56cf17e16c03614689af6d35"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:47:02.542308Z","signature_b64":"+5RwM2tW//VTD5xH5YSC/R5pQXvCyCf7ecPYQU5FfSuFFloM1ptNLnJc5HdlNz9Y6vY4wbgNTCQUFn3hbOmBAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"578c187064717e8880d693cef26149174da084617cdd67bee8aa708ff6bc0e17","last_reissued_at":"2026-07-05T00:47:02.541837Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:47:02.541837Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Multi-View Learning for Vision-and-Language Navigation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","cs.LG"],"primary_cat":"cs.CL","authors_text":"Chunyuan Li, Jianfeng Gao, Noah A. Smith, Qiaolin Xia, Xiujun Li, Yejin Choi, Yonatan Bisk, Zhifang Sui","submitted_at":"2020-03-02T13:07:46Z","abstract_excerpt":"Learning to navigate in a visual environment following natural language instructions is a challenging task because natural language instructions are highly variable, ambiguous, and under-specified. In this paper, we present a novel training paradigm, Learn from EveryOne (LEO), which leverages multiple instructions (as different views) for the same trajectory to resolve language ambiguity and improve generalization. By sharing parameters across instructions, our approach learns more effectively from limited training data and generalizes better in unseen environments. On the recent Room-to-Room "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2003.00857","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2003.00857/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2003.00857","created_at":"2026-07-05T00:47:02.541894+00:00"},{"alias_kind":"arxiv_version","alias_value":"2003.00857v3","created_at":"2026-07-05T00:47:02.541894+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2003.00857","created_at":"2026-07-05T00:47:02.541894+00:00"},{"alias_kind":"pith_short_12","alias_value":"K6GBQ4DEOF7I","created_at":"2026-07-05T00:47:02.541894+00:00"},{"alias_kind":"pith_short_16","alias_value":"K6GBQ4DEOF7IRAGW","created_at":"2026-07-05T00:47:02.541894+00:00"},{"alias_kind":"pith_short_8","alias_value":"K6GBQ4DE","created_at":"2026-07-05T00:47:02.541894+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.13328","citing_title":"What Limits Vision-and-Language Navigation ?","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/K6GBQ4DEOF7IRAGWSPHPEYKJC5","json":"https://pith.science/pith/K6GBQ4DEOF7IRAGWSPHPEYKJC5.json","graph_json":"https://pith.science/api/pith-number/K6GBQ4DEOF7IRAGWSPHPEYKJC5/graph.json","events_json":"https://pith.science/api/pith-number/K6GBQ4DEOF7IRAGWSPHPEYKJC5/events.json","paper":"https://pith.science/paper/K6GBQ4DE"},"agent_actions":{"view_html":"https://pith.science/pith/K6GBQ4DEOF7IRAGWSPHPEYKJC5","download_json":"https://pith.science/pith/K6GBQ4DEOF7IRAGWSPHPEYKJC5.json","view_paper":"https://pith.science/paper/K6GBQ4DE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2003.00857&json=true","fetch_graph":"https://pith.science/api/pith-number/K6GBQ4DEOF7IRAGWSPHPEYKJC5/graph.json","fetch_events":"https://pith.science/api/pith-number/K6GBQ4DEOF7IRAGWSPHPEYKJC5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/K6GBQ4DEOF7IRAGWSPHPEYKJC5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/K6GBQ4DEOF7IRAGWSPHPEYKJC5/action/storage_attestation","attest_author":"https://pith.science/pith/K6GBQ4DEOF7IRAGWSPHPEYKJC5/action/author_attestation","sign_citation":"https://pith.science/pith/K6GBQ4DEOF7IRAGWSPHPEYKJC5/action/citation_signature","submit_replication":"https://pith.science/pith/K6GBQ4DEOF7IRAGWSPHPEYKJC5/action/replication_record"}},"created_at":"2026-07-05T00:47:02.541894+00:00","updated_at":"2026-07-05T00:47:02.541894+00:00"}