{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:H2NMJW3LOWCEWBWZXSIZMLKJMC","short_pith_number":"pith:H2NMJW3L","schema_version":"1.0","canonical_sha256":"3e9ac4db6b75844b06d9bc91962d4960b0bf992afe08021549ce9e157fa7fcbc","source":{"kind":"arxiv","id":"2308.07997","version":1},"attestation_state":"computed","paper":{"title":"$A^2$Nav: Action-Aware Zero-Shot Robot Navigation by Exploiting Vision-and-Language Ability of Foundation Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.CV","authors_text":"Chuang Gan, Gaowen Liu, Hongyan Zhi, Mingkui Tan, Peihao Chen, Runhao Zeng, Thomas H. Li, Xinyu Sun","submitted_at":"2023-08-15T19:01:19Z","abstract_excerpt":"We study the task of zero-shot vision-and-language navigation (ZS-VLN), a practical yet challenging problem in which an agent learns to navigate following a path described by language instructions without requiring any path-instruction annotation data. Normally, the instructions have complex grammatical structures and often contain various action descriptions (e.g., \"proceed beyond\", \"depart from\"). How to correctly understand and execute these action demands is a critical problem, and the absence of annotated data makes it even more challenging. Note that a well-educated human being can easil"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.07997","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-08-15T19:01:19Z","cross_cats_sorted":["cs.RO"],"title_canon_sha256":"8d81b7abc96a0ee721c606703aa931beafe369570e390df37a0a70250f919075","abstract_canon_sha256":"26797976c2310e50b72bf36791713d1da7351565dfdf0cb805f0aeb441ad3357"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:41:52.847903Z","signature_b64":"4RB7AWWF4Zm3wwWFA+kj31vVVWYEAG/rs2CPLKQMOplJGjBRp8kkLBlbcz7mwkV48XLesXc1ZQ3zR/15DIzOBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3e9ac4db6b75844b06d9bc91962d4960b0bf992afe08021549ce9e157fa7fcbc","last_reissued_at":"2026-07-05T06:41:52.847423Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:41:52.847423Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"$A^2$Nav: Action-Aware Zero-Shot Robot Navigation by Exploiting Vision-and-Language Ability of Foundation Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.CV","authors_text":"Chuang Gan, Gaowen Liu, Hongyan Zhi, Mingkui Tan, Peihao Chen, Runhao Zeng, Thomas H. Li, Xinyu Sun","submitted_at":"2023-08-15T19:01:19Z","abstract_excerpt":"We study the task of zero-shot vision-and-language navigation (ZS-VLN), a practical yet challenging problem in which an agent learns to navigate following a path described by language instructions without requiring any path-instruction annotation data. Normally, the instructions have complex grammatical structures and often contain various action descriptions (e.g., \"proceed beyond\", \"depart from\"). How to correctly understand and execute these action demands is a critical problem, and the absence of annotated data makes it even more challenging. Note that a well-educated human being can easil"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.07997","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.07997/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.07997","created_at":"2026-07-05T06:41:52.847484+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.07997v1","created_at":"2026-07-05T06:41:52.847484+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.07997","created_at":"2026-07-05T06:41:52.847484+00:00"},{"alias_kind":"pith_short_12","alias_value":"H2NMJW3LOWCE","created_at":"2026-07-05T06:41:52.847484+00:00"},{"alias_kind":"pith_short_16","alias_value":"H2NMJW3LOWCEWBWZ","created_at":"2026-07-05T06:41:52.847484+00:00"},{"alias_kind":"pith_short_8","alias_value":"H2NMJW3L","created_at":"2026-07-05T06:41:52.847484+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.10577","citing_title":"AgenticNav: Zero-Shot Vision-and-Language Navigation as a Tool-Calling Harness","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00095","citing_title":"Bridging the 2D-3D Gap: A Hierarchical Semantic-Geometric Map for Vision Language Navigation","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22816","citing_title":"AwareVLN: Reasoning with Self-awareness for Vision-Language Navigation","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2402.15852","citing_title":"NaVid: Video-based VLM Plans the Next Step for Vision-and-Language Navigation","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2511.17097","citing_title":"Progress-Think: Semantic Progress Reasoning for Vision-Language Navigation","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2412.06224","citing_title":"Uni-NaVid: A Video-based Vision-Language-Action Model for Unifying Embodied Navigation Tasks","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16298","citing_title":"FineCog-Nav: Integrating Fine-grained Cognitive Modules for Zero-shot Multimodal UAV Navigation","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/H2NMJW3LOWCEWBWZXSIZMLKJMC","json":"https://pith.science/pith/H2NMJW3LOWCEWBWZXSIZMLKJMC.json","graph_json":"https://pith.science/api/pith-number/H2NMJW3LOWCEWBWZXSIZMLKJMC/graph.json","events_json":"https://pith.science/api/pith-number/H2NMJW3LOWCEWBWZXSIZMLKJMC/events.json","paper":"https://pith.science/paper/H2NMJW3L"},"agent_actions":{"view_html":"https://pith.science/pith/H2NMJW3LOWCEWBWZXSIZMLKJMC","download_json":"https://pith.science/pith/H2NMJW3LOWCEWBWZXSIZMLKJMC.json","view_paper":"https://pith.science/paper/H2NMJW3L","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.07997&json=true","fetch_graph":"https://pith.science/api/pith-number/H2NMJW3LOWCEWBWZXSIZMLKJMC/graph.json","fetch_events":"https://pith.science/api/pith-number/H2NMJW3LOWCEWBWZXSIZMLKJMC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/H2NMJW3LOWCEWBWZXSIZMLKJMC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/H2NMJW3LOWCEWBWZXSIZMLKJMC/action/storage_attestation","attest_author":"https://pith.science/pith/H2NMJW3LOWCEWBWZXSIZMLKJMC/action/author_attestation","sign_citation":"https://pith.science/pith/H2NMJW3LOWCEWBWZXSIZMLKJMC/action/citation_signature","submit_replication":"https://pith.science/pith/H2NMJW3LOWCEWBWZXSIZMLKJMC/action/replication_record"}},"created_at":"2026-07-05T06:41:52.847484+00:00","updated_at":"2026-07-05T06:41:52.847484+00:00"}