{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:WC25PC7HEWTWQU3LULWMA3CQGW","short_pith_number":"pith:WC25PC7H","schema_version":"1.0","canonical_sha256":"b0b5d78be725a768536ba2ecc06c50359c7e4409fde1e6caf5aec948f445625d","source":{"kind":"arxiv","id":"2408.05090","version":1},"attestation_state":"computed","paper":{"title":"Loc4Plan: Locating Before Planning for Outdoor Vision and Language Navigation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.MM"],"primary_cat":"cs.CV","authors_text":"Huilin Tian, Jingke Meng, Junkai Yan, Wei-Shi Zheng, Yuan-Ming Li, Yunong Zhang","submitted_at":"2024-08-09T14:31:09Z","abstract_excerpt":"Vision and Language Navigation (VLN) is a challenging task that requires agents to understand instructions and navigate to the destination in a visual environment.One of the key challenges in outdoor VLN is keeping track of which part of the instruction was completed. To alleviate this problem, previous works mainly focus on grounding the natural language to the visual input, but neglecting the crucial role of the agent's spatial position information in the grounding process. In this work, we first explore the substantial effect of spatial position locating on the grounding of outdoor VLN, dra"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.05090","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-08-09T14:31:09Z","cross_cats_sorted":["cs.MM"],"title_canon_sha256":"063d6d267e17d36e9d5360abcc881c77ac1ba7829852dc11870beeb79141b4a5","abstract_canon_sha256":"987692e57816976b25a597018ee87815411e804410824ef946aad494e1ab44ad"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:53:56.635292Z","signature_b64":"9WMz5ex2P2d9I4cSNfDpAPUXyrDAqynIR2BOeOBrDkgWHZ0gj0u50By44CTaT4Fmn8zV4tFaHU8aTQAdWzDRCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b0b5d78be725a768536ba2ecc06c50359c7e4409fde1e6caf5aec948f445625d","last_reissued_at":"2026-07-05T08:53:56.634835Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:53:56.634835Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Loc4Plan: Locating Before Planning for Outdoor Vision and Language Navigation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.MM"],"primary_cat":"cs.CV","authors_text":"Huilin Tian, Jingke Meng, Junkai Yan, Wei-Shi Zheng, Yuan-Ming Li, Yunong Zhang","submitted_at":"2024-08-09T14:31:09Z","abstract_excerpt":"Vision and Language Navigation (VLN) is a challenging task that requires agents to understand instructions and navigate to the destination in a visual environment.One of the key challenges in outdoor VLN is keeping track of which part of the instruction was completed. To alleviate this problem, previous works mainly focus on grounding the natural language to the visual input, but neglecting the crucial role of the agent's spatial position information in the grounding process. In this work, we first explore the substantial effect of spatial position locating on the grounding of outdoor VLN, dra"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.05090","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.05090/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.05090","created_at":"2026-07-05T08:53:56.634902+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.05090v1","created_at":"2026-07-05T08:53:56.634902+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.05090","created_at":"2026-07-05T08:53:56.634902+00:00"},{"alias_kind":"pith_short_12","alias_value":"WC25PC7HEWTW","created_at":"2026-07-05T08:53:56.634902+00:00"},{"alias_kind":"pith_short_16","alias_value":"WC25PC7HEWTWQU3L","created_at":"2026-07-05T08:53:56.634902+00:00"},{"alias_kind":"pith_short_8","alias_value":"WC25PC7H","created_at":"2026-07-05T08:53:56.634902+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.17007","citing_title":"Where am I? Cross-View Geo-localization with Natural Language Descriptions","ref_index":37,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WC25PC7HEWTWQU3LULWMA3CQGW","json":"https://pith.science/pith/WC25PC7HEWTWQU3LULWMA3CQGW.json","graph_json":"https://pith.science/api/pith-number/WC25PC7HEWTWQU3LULWMA3CQGW/graph.json","events_json":"https://pith.science/api/pith-number/WC25PC7HEWTWQU3LULWMA3CQGW/events.json","paper":"https://pith.science/paper/WC25PC7H"},"agent_actions":{"view_html":"https://pith.science/pith/WC25PC7HEWTWQU3LULWMA3CQGW","download_json":"https://pith.science/pith/WC25PC7HEWTWQU3LULWMA3CQGW.json","view_paper":"https://pith.science/paper/WC25PC7H","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.05090&json=true","fetch_graph":"https://pith.science/api/pith-number/WC25PC7HEWTWQU3LULWMA3CQGW/graph.json","fetch_events":"https://pith.science/api/pith-number/WC25PC7HEWTWQU3LULWMA3CQGW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WC25PC7HEWTWQU3LULWMA3CQGW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WC25PC7HEWTWQU3LULWMA3CQGW/action/storage_attestation","attest_author":"https://pith.science/pith/WC25PC7HEWTWQU3LULWMA3CQGW/action/author_attestation","sign_citation":"https://pith.science/pith/WC25PC7HEWTWQU3LULWMA3CQGW/action/citation_signature","submit_replication":"https://pith.science/pith/WC25PC7HEWTWQU3LULWMA3CQGW/action/replication_record"}},"created_at":"2026-07-05T08:53:56.634902+00:00","updated_at":"2026-07-05T08:53:56.634902+00:00"}