{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:L2J47U2BELD3RL2MR6VAHVRI2A","short_pith_number":"pith:L2J47U2B","schema_version":"1.0","canonical_sha256":"5e93cfd34122c7b8af4c8faa03d628d0096ff0f9faf9436f3ecf33ab736e5c20","source":{"kind":"arxiv","id":"2108.09105","version":1},"attestation_state":"computed","paper":{"title":"Airbert: In-domain Pretraining for Vision-and-Language Navigation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.HC","cs.LG"],"primary_cat":"cs.CV","authors_text":"Cordelia Schmid, Ivan Laptev, Makarand Tapaswi, Pierre-Louis Guhur, Shizhe Chen","submitted_at":"2021-08-20T10:58:09Z","abstract_excerpt":"Vision-and-language navigation (VLN) aims to enable embodied agents to navigate in realistic environments using natural language instructions. Given the scarcity of domain-specific training data and the high diversity of image and language inputs, the generalization of VLN agents to unseen environments remains challenging. Recent methods explore pretraining to improve generalization, however, the use of generic image-caption datasets or existing small-scale VLN environments is suboptimal and results in limited improvements. In this work, we introduce BnB, a large-scale and diverse in-domain VL"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2108.09105","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2021-08-20T10:58:09Z","cross_cats_sorted":["cs.AI","cs.CL","cs.HC","cs.LG"],"title_canon_sha256":"120ee1e2fa10fcdd8bcb1c9d51af51d95cac73a718f69144e4ead2b95d0598d6","abstract_canon_sha256":"3c677b0d5b6943992def75c4f6be0e7996e9724e1bf23a6acc245e7ade3ae81e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:07:34.208297Z","signature_b64":"Adofx22TumakbV7yX4NOOi5rnN9EqXtyTp7pmKlhayzzozltA+xb5mS8i0xHNKezK7iH68E9//t00dmiNQV0Dw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5e93cfd34122c7b8af4c8faa03d628d0096ff0f9faf9436f3ecf33ab736e5c20","last_reissued_at":"2026-07-05T03:07:34.207825Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:07:34.207825Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Airbert: In-domain Pretraining for Vision-and-Language Navigation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.HC","cs.LG"],"primary_cat":"cs.CV","authors_text":"Cordelia Schmid, Ivan Laptev, Makarand Tapaswi, Pierre-Louis Guhur, Shizhe Chen","submitted_at":"2021-08-20T10:58:09Z","abstract_excerpt":"Vision-and-language navigation (VLN) aims to enable embodied agents to navigate in realistic environments using natural language instructions. Given the scarcity of domain-specific training data and the high diversity of image and language inputs, the generalization of VLN agents to unseen environments remains challenging. Recent methods explore pretraining to improve generalization, however, the use of generic image-caption datasets or existing small-scale VLN environments is suboptimal and results in limited improvements. In this work, we introduce BnB, a large-scale and diverse in-domain VL"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2108.09105","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2108.09105/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2108.09105","created_at":"2026-07-05T03:07:34.207881+00:00"},{"alias_kind":"arxiv_version","alias_value":"2108.09105v1","created_at":"2026-07-05T03:07:34.207881+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2108.09105","created_at":"2026-07-05T03:07:34.207881+00:00"},{"alias_kind":"pith_short_12","alias_value":"L2J47U2BELD3","created_at":"2026-07-05T03:07:34.207881+00:00"},{"alias_kind":"pith_short_16","alias_value":"L2J47U2BELD3RL2M","created_at":"2026-07-05T03:07:34.207881+00:00"},{"alias_kind":"pith_short_8","alias_value":"L2J47U2B","created_at":"2026-07-05T03:07:34.207881+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.16654","citing_title":"MSNav: Zero-Shot Vision-and-Language Navigation with Dynamic Memory and LLM Spatial Reasoning","ref_index":20,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/L2J47U2BELD3RL2MR6VAHVRI2A","json":"https://pith.science/pith/L2J47U2BELD3RL2MR6VAHVRI2A.json","graph_json":"https://pith.science/api/pith-number/L2J47U2BELD3RL2MR6VAHVRI2A/graph.json","events_json":"https://pith.science/api/pith-number/L2J47U2BELD3RL2MR6VAHVRI2A/events.json","paper":"https://pith.science/paper/L2J47U2B"},"agent_actions":{"view_html":"https://pith.science/pith/L2J47U2BELD3RL2MR6VAHVRI2A","download_json":"https://pith.science/pith/L2J47U2BELD3RL2MR6VAHVRI2A.json","view_paper":"https://pith.science/paper/L2J47U2B","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2108.09105&json=true","fetch_graph":"https://pith.science/api/pith-number/L2J47U2BELD3RL2MR6VAHVRI2A/graph.json","fetch_events":"https://pith.science/api/pith-number/L2J47U2BELD3RL2MR6VAHVRI2A/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/L2J47U2BELD3RL2MR6VAHVRI2A/action/timestamp_anchor","attest_storage":"https://pith.science/pith/L2J47U2BELD3RL2MR6VAHVRI2A/action/storage_attestation","attest_author":"https://pith.science/pith/L2J47U2BELD3RL2MR6VAHVRI2A/action/author_attestation","sign_citation":"https://pith.science/pith/L2J47U2BELD3RL2MR6VAHVRI2A/action/citation_signature","submit_replication":"https://pith.science/pith/L2J47U2BELD3RL2MR6VAHVRI2A/action/replication_record"}},"created_at":"2026-07-05T03:07:34.207881+00:00","updated_at":"2026-07-05T03:07:34.207881+00:00"}