{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:YOU2MYV2HJ2IZYVVPUFPVFWXK5","short_pith_number":"pith:YOU2MYV2","schema_version":"1.0","canonical_sha256":"c3a9a662ba3a748ce2b57d0afa96d7577fc62b0f38af028c20242f956b33ca26","source":{"kind":"arxiv","id":"2410.08500","version":3},"attestation_state":"computed","paper":{"title":"Exploring Spatial Representation to Enhance LLM Reasoning in Aerial Vision-Language Navigation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.RO","authors_text":"Bin Zhao, Dong Wang, Linglin Jing, Pengfei Han, Yunpeng Gao, Zhigang Wang","submitted_at":"2024-10-11T03:54:48Z","abstract_excerpt":"Aerial Vision-and-Language Navigation (VLN) is a novel task enabling Unmanned Aerial Vehicles (UAVs) to navigate in outdoor environments through natural language instructions and visual cues. However, it remains challenging due to the complex spatial relationships in aerial scenes.In this paper, we propose a training-free, zero-shot framework for aerial VLN tasks, where the large language model (LLM) is leveraged as the agent for action prediction. Specifically, we develop a novel Semantic-Topo-Metric Representation (STMR) to enhance the spatial reasoning capabilities of LLMs. This is achieved"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.08500","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2024-10-11T03:54:48Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"9d98fff7ad442e33b305ea4aa938e9ca0f56312aabe3ec99fb7a47a72d9e4719","abstract_canon_sha256":"321ea4098814e34fc3eacf618e2cd1ba59ec2c13cf7a873ab60f6830bef7accb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:51:18.188346Z","signature_b64":"Etoq+m9lmepV2gDKFei6i0jYVx8GR+BdDO47j28U88u77dN5+cBeeUHcEXhYdQlxfDh3Pkgyeijfc8rZewnmDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c3a9a662ba3a748ce2b57d0afa96d7577fc62b0f38af028c20242f956b33ca26","last_reissued_at":"2026-07-05T11:51:18.187821Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:51:18.187821Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Exploring Spatial Representation to Enhance LLM Reasoning in Aerial Vision-Language Navigation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.RO","authors_text":"Bin Zhao, Dong Wang, Linglin Jing, Pengfei Han, Yunpeng Gao, Zhigang Wang","submitted_at":"2024-10-11T03:54:48Z","abstract_excerpt":"Aerial Vision-and-Language Navigation (VLN) is a novel task enabling Unmanned Aerial Vehicles (UAVs) to navigate in outdoor environments through natural language instructions and visual cues. However, it remains challenging due to the complex spatial relationships in aerial scenes.In this paper, we propose a training-free, zero-shot framework for aerial VLN tasks, where the large language model (LLM) is leveraged as the agent for action prediction. Specifically, we develop a novel Semantic-Topo-Metric Representation (STMR) to enhance the spatial reasoning capabilities of LLMs. This is achieved"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.08500","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.08500/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.08500","created_at":"2026-07-05T11:51:18.187881+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.08500v3","created_at":"2026-07-05T11:51:18.187881+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.08500","created_at":"2026-07-05T11:51:18.187881+00:00"},{"alias_kind":"pith_short_12","alias_value":"YOU2MYV2HJ2I","created_at":"2026-07-05T11:51:18.187881+00:00"},{"alias_kind":"pith_short_16","alias_value":"YOU2MYV2HJ2IZYVV","created_at":"2026-07-05T11:51:18.187881+00:00"},{"alias_kind":"pith_short_8","alias_value":"YOU2MYV2","created_at":"2026-07-05T11:51:18.187881+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.28397","citing_title":"CLOSER-VLN: Closed-Loop Self-Verified Retrieval-Augmented Reasoning for Aerial Vision-Language Navigation","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2512.08639","citing_title":"Aerial Vision-Language Navigation with a Unified Framework for Spatial, Temporal and Embodied Reasoning","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07973","citing_title":"How Far Are Large Multimodal Models from Human-Level Spatial Action? A Benchmark for Goal-Oriented Embodied Navigation in Urban Airspace","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08883","citing_title":"HTNav: A Hybrid Navigation Framework with Tiered Structure for Urban Aerial Vision-and-Language Navigation","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07705","citing_title":"Vision-Language Navigation for Aerial Robots: Towards the Era of Large Language Models","ref_index":91,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13654","citing_title":"Vision-and-Language Navigation for UAVs: Progress, Challenges, and a Research Roadmap","ref_index":294,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16298","citing_title":"FineCog-Nav: Integrating Fine-grained Cognitive Modules for Zero-shot Multimodal UAV Navigation","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17190","citing_title":"LookasideVLN: Direction-Aware Aerial Vision-and-Language Navigation","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YOU2MYV2HJ2IZYVVPUFPVFWXK5","json":"https://pith.science/pith/YOU2MYV2HJ2IZYVVPUFPVFWXK5.json","graph_json":"https://pith.science/api/pith-number/YOU2MYV2HJ2IZYVVPUFPVFWXK5/graph.json","events_json":"https://pith.science/api/pith-number/YOU2MYV2HJ2IZYVVPUFPVFWXK5/events.json","paper":"https://pith.science/paper/YOU2MYV2"},"agent_actions":{"view_html":"https://pith.science/pith/YOU2MYV2HJ2IZYVVPUFPVFWXK5","download_json":"https://pith.science/pith/YOU2MYV2HJ2IZYVVPUFPVFWXK5.json","view_paper":"https://pith.science/paper/YOU2MYV2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.08500&json=true","fetch_graph":"https://pith.science/api/pith-number/YOU2MYV2HJ2IZYVVPUFPVFWXK5/graph.json","fetch_events":"https://pith.science/api/pith-number/YOU2MYV2HJ2IZYVVPUFPVFWXK5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YOU2MYV2HJ2IZYVVPUFPVFWXK5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YOU2MYV2HJ2IZYVVPUFPVFWXK5/action/storage_attestation","attest_author":"https://pith.science/pith/YOU2MYV2HJ2IZYVVPUFPVFWXK5/action/author_attestation","sign_citation":"https://pith.science/pith/YOU2MYV2HJ2IZYVVPUFPVFWXK5/action/citation_signature","submit_replication":"https://pith.science/pith/YOU2MYV2HJ2IZYVVPUFPVFWXK5/action/replication_record"}},"created_at":"2026-07-05T11:51:18.187881+00:00","updated_at":"2026-07-05T11:51:18.187881+00:00"}