{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:CFBUFHUVY5AGYPKKKPU3TWWZQD","short_pith_number":"pith:CFBUFHUV","schema_version":"1.0","canonical_sha256":"1143429e95c7406c3d4a53e9b9dad980c671332fd5eb7eddac918ebe88c1314c","source":{"kind":"arxiv","id":"2411.11394","version":1},"attestation_state":"computed","paper":{"title":"InstruGen: Automatic Instruction Generation for Vision-and-Language Navigation Via Large Multimodal Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Jianqin Yin, Jiazhao Zhang, Peiyang Li, Rongtao Xu, Xiaodan Liang, Yu Yan","submitted_at":"2024-11-18T09:11:48Z","abstract_excerpt":"Recent research on Vision-and-Language Navigation (VLN) indicates that agents suffer from poor generalization in unseen environments due to the lack of realistic training environments and high-quality path-instruction pairs. Most existing methods for constructing realistic navigation scenes have high costs, and the extension of instructions mainly relies on predefined templates or rules, lacking adaptability. To alleviate the issue, we propose InstruGen, a VLN path-instruction pairs generation paradigm. Specifically, we use YouTube house tour videos as realistic navigation scenes and leverage "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.11394","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2024-11-18T09:11:48Z","cross_cats_sorted":[],"title_canon_sha256":"e4e0cda7888c7a3acf858718f8c565e39212adf9e0eb2a9353ae64e55571074a","abstract_canon_sha256":"df901ee3c1b7c759d35f3efb75df59ebfcb507e847e80e718ef5b6c72f2ec769"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:36:55.761084Z","signature_b64":"vl1/+DW+rkwXxdmxZFd6y7TUff+DzyDN7Qn4pH7t+DMVZdsrxR8IfnMGtWclHAlC0nno+csrtIccTw6O0tjYCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1143429e95c7406c3d4a53e9b9dad980c671332fd5eb7eddac918ebe88c1314c","last_reissued_at":"2026-07-05T09:36:55.760596Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:36:55.760596Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"InstruGen: Automatic Instruction Generation for Vision-and-Language Navigation Via Large Multimodal Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Jianqin Yin, Jiazhao Zhang, Peiyang Li, Rongtao Xu, Xiaodan Liang, Yu Yan","submitted_at":"2024-11-18T09:11:48Z","abstract_excerpt":"Recent research on Vision-and-Language Navigation (VLN) indicates that agents suffer from poor generalization in unseen environments due to the lack of realistic training environments and high-quality path-instruction pairs. Most existing methods for constructing realistic navigation scenes have high costs, and the extension of instructions mainly relies on predefined templates or rules, lacking adaptability. To alleviate the issue, we propose InstruGen, a VLN path-instruction pairs generation paradigm. Specifically, we use YouTube house tour videos as realistic navigation scenes and leverage "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.11394","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.11394/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.11394","created_at":"2026-07-05T09:36:55.760651+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.11394v1","created_at":"2026-07-05T09:36:55.760651+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.11394","created_at":"2026-07-05T09:36:55.760651+00:00"},{"alias_kind":"pith_short_12","alias_value":"CFBUFHUVY5AG","created_at":"2026-07-05T09:36:55.760651+00:00"},{"alias_kind":"pith_short_16","alias_value":"CFBUFHUVY5AGYPKK","created_at":"2026-07-05T09:36:55.760651+00:00"},{"alias_kind":"pith_short_8","alias_value":"CFBUFHUV","created_at":"2026-07-05T09:36:55.760651+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01754","citing_title":"Path-level Hindsight Instructions for Semantic Exploration in Vision-Language Navigation","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2508.09547","citing_title":"GoViG: Goal-Conditioned Visual Navigation Instruction Generation via Multimodal Reasoning","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CFBUFHUVY5AGYPKKKPU3TWWZQD","json":"https://pith.science/pith/CFBUFHUVY5AGYPKKKPU3TWWZQD.json","graph_json":"https://pith.science/api/pith-number/CFBUFHUVY5AGYPKKKPU3TWWZQD/graph.json","events_json":"https://pith.science/api/pith-number/CFBUFHUVY5AGYPKKKPU3TWWZQD/events.json","paper":"https://pith.science/paper/CFBUFHUV"},"agent_actions":{"view_html":"https://pith.science/pith/CFBUFHUVY5AGYPKKKPU3TWWZQD","download_json":"https://pith.science/pith/CFBUFHUVY5AGYPKKKPU3TWWZQD.json","view_paper":"https://pith.science/paper/CFBUFHUV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.11394&json=true","fetch_graph":"https://pith.science/api/pith-number/CFBUFHUVY5AGYPKKKPU3TWWZQD/graph.json","fetch_events":"https://pith.science/api/pith-number/CFBUFHUVY5AGYPKKKPU3TWWZQD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CFBUFHUVY5AGYPKKKPU3TWWZQD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CFBUFHUVY5AGYPKKKPU3TWWZQD/action/storage_attestation","attest_author":"https://pith.science/pith/CFBUFHUVY5AGYPKKKPU3TWWZQD/action/author_attestation","sign_citation":"https://pith.science/pith/CFBUFHUVY5AGYPKKKPU3TWWZQD/action/citation_signature","submit_replication":"https://pith.science/pith/CFBUFHUVY5AGYPKKKPU3TWWZQD/action/replication_record"}},"created_at":"2026-07-05T09:36:55.760651+00:00","updated_at":"2026-07-05T09:36:55.760651+00:00"}