{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:22XWZNLVM5G6DJICMYPTEU2AJU","short_pith_number":"pith:22XWZNLV","schema_version":"1.0","canonical_sha256":"d6af6cb575674de1a502661f3253404d1146e58614b20c3f4da44c00f4d950ee","source":{"kind":"arxiv","id":"2411.08579","version":1},"attestation_state":"computed","paper":{"title":"NavAgent: Multi-scale Urban Street View Fusion For UAV Embodied Vision-and-Language Navigation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.CV","authors_text":"Fanglong Yao, Guangluan Xu, Kun Fu, Xian Sun, Youzhi Liu, Yuanchang Yue","submitted_at":"2024-11-13T12:51:49Z","abstract_excerpt":"Vision-and-Language Navigation (VLN), as a widely discussed research direction in embodied intelligence, aims to enable embodied agents to navigate in complicated visual environments through natural language commands. Most existing VLN methods focus on indoor ground robot scenarios. However, when applied to UAV VLN in outdoor urban scenes, it faces two significant challenges. First, urban scenes contain numerous objects, which makes it challenging to match fine-grained landmarks in images with complex textual descriptions of these landmarks. Second, overall environmental information encompasse"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.08579","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-11-13T12:51:49Z","cross_cats_sorted":["cs.RO"],"title_canon_sha256":"ea0bfb108431c9288aa1b16ce737965518bbea8dd97446535f0c34c40953f209","abstract_canon_sha256":"e70a954a20089b614e9a21744a42ff39db19efa9424ea04c9e00492ffac2916f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:34:59.272413Z","signature_b64":"A8F/u9uzXW345UUSBRElJ9sCt0ZOzj+1tA+nqtwbSD4ZmLNRtX3mbKA695N1JAMjo0HAwJomVZpx7shz4la1DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d6af6cb575674de1a502661f3253404d1146e58614b20c3f4da44c00f4d950ee","last_reissued_at":"2026-07-05T09:34:59.271937Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:34:59.271937Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"NavAgent: Multi-scale Urban Street View Fusion For UAV Embodied Vision-and-Language Navigation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.CV","authors_text":"Fanglong Yao, Guangluan Xu, Kun Fu, Xian Sun, Youzhi Liu, Yuanchang Yue","submitted_at":"2024-11-13T12:51:49Z","abstract_excerpt":"Vision-and-Language Navigation (VLN), as a widely discussed research direction in embodied intelligence, aims to enable embodied agents to navigate in complicated visual environments through natural language commands. Most existing VLN methods focus on indoor ground robot scenarios. However, when applied to UAV VLN in outdoor urban scenes, it faces two significant challenges. First, urban scenes contain numerous objects, which makes it challenging to match fine-grained landmarks in images with complex textual descriptions of these landmarks. Second, overall environmental information encompasse"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.08579","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.08579/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.08579","created_at":"2026-07-05T09:34:59.271990+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.08579v1","created_at":"2026-07-05T09:34:59.271990+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.08579","created_at":"2026-07-05T09:34:59.271990+00:00"},{"alias_kind":"pith_short_12","alias_value":"22XWZNLVM5G6","created_at":"2026-07-05T09:34:59.271990+00:00"},{"alias_kind":"pith_short_16","alias_value":"22XWZNLVM5G6DJIC","created_at":"2026-07-05T09:34:59.271990+00:00"},{"alias_kind":"pith_short_8","alias_value":"22XWZNLV","created_at":"2026-07-05T09:34:59.271990+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20045","citing_title":"See-and-Reach: Precise Vision-Language Navigation for UAVs within the Field of View","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31654","citing_title":"DynFly: Dynamic-Aware Continuous Trajectory Generation for UAV Vision-Language Navigation in Urban Environments","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31654","citing_title":"DynFly: Dynamic-Aware Continuous Trajectory Generation for UAV Vision-Language Navigation in Urban Environments","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31467","citing_title":"AeroVerse-SatAgent: UAV-Satellite Collaborative Spatial Reasoning Inspired by the Dual Visual Pathway Theory of Cognitive Neuroscience","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00104","citing_title":"PEACE: A Planner-Executor Agent with Constraint Enforcement for UAVs","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07973","citing_title":"How Far Are Large Multimodal Models from Human-Level Spatial Action? A Benchmark for Goal-Oriented Embodied Navigation in Urban Airspace","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08883","citing_title":"HTNav: A Hybrid Navigation Framework with Tiered Structure for Urban Aerial Vision-and-Language Navigation","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07705","citing_title":"Vision-Language Navigation for Aerial Robots: Towards the Era of Large Language Models","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13654","citing_title":"Vision-and-Language Navigation for UAVs: Progress, Challenges, and a Research Roadmap","ref_index":121,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16298","citing_title":"FineCog-Nav: Integrating Fine-grained Cognitive Modules for Zero-shot Multimodal UAV Navigation","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16993","citing_title":"Rule-VLN: Bridging Perception and Compliance via Semantic Reasoning and Geometric Rectification","ref_index":33,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/22XWZNLVM5G6DJICMYPTEU2AJU","json":"https://pith.science/pith/22XWZNLVM5G6DJICMYPTEU2AJU.json","graph_json":"https://pith.science/api/pith-number/22XWZNLVM5G6DJICMYPTEU2AJU/graph.json","events_json":"https://pith.science/api/pith-number/22XWZNLVM5G6DJICMYPTEU2AJU/events.json","paper":"https://pith.science/paper/22XWZNLV"},"agent_actions":{"view_html":"https://pith.science/pith/22XWZNLVM5G6DJICMYPTEU2AJU","download_json":"https://pith.science/pith/22XWZNLVM5G6DJICMYPTEU2AJU.json","view_paper":"https://pith.science/paper/22XWZNLV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.08579&json=true","fetch_graph":"https://pith.science/api/pith-number/22XWZNLVM5G6DJICMYPTEU2AJU/graph.json","fetch_events":"https://pith.science/api/pith-number/22XWZNLVM5G6DJICMYPTEU2AJU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/22XWZNLVM5G6DJICMYPTEU2AJU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/22XWZNLVM5G6DJICMYPTEU2AJU/action/storage_attestation","attest_author":"https://pith.science/pith/22XWZNLVM5G6DJICMYPTEU2AJU/action/author_attestation","sign_citation":"https://pith.science/pith/22XWZNLVM5G6DJICMYPTEU2AJU/action/citation_signature","submit_replication":"https://pith.science/pith/22XWZNLVM5G6DJICMYPTEU2AJU/action/replication_record"}},"created_at":"2026-07-05T09:34:59.271990+00:00","updated_at":"2026-07-05T09:34:59.271990+00:00"}