{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:FXZGQA4UVVWM6NNFOXQFSNVKE4","short_pith_number":"pith:FXZGQA4U","schema_version":"1.0","canonical_sha256":"2df2680394ad6ccf35a575e05936aa270bab857dba126b18475ddf2ca2af8cc1","source":{"kind":"arxiv","id":"2410.09874","version":1},"attestation_state":"computed","paper":{"title":"ImagineNav: Prompting Vision-Language Models as Embodied Navigator through Scene Imagination","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Likun Tang, Teng Wang, Wenzhe Cai, Xinxin Zhao","submitted_at":"2024-10-13T15:31:31Z","abstract_excerpt":"Visual navigation is an essential skill for home-assistance robots, providing the object-searching ability to accomplish long-horizon daily tasks. Many recent approaches use Large Language Models (LLMs) for commonsense inference to improve exploration efficiency. However, the planning process of LLMs is limited within texts and it is difficult to represent the spatial occupancy and geometry layout only by texts. Both are important for making rational navigation decisions. In this work, we seek to unleash the spatial perception and planning ability of Vision-Language Models (VLMs), and explore "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.09874","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2024-10-13T15:31:31Z","cross_cats_sorted":[],"title_canon_sha256":"0ef65478ddc610bcd90d679f022e13b8672c108653709780f2f1e64637b94b32","abstract_canon_sha256":"8c5468f9927ad006d7c1572ce456e587e613668384b594f2f9d0abfdc1dde5be"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:20:07.411654Z","signature_b64":"Nz/PgSTne5iqPjkEEWCVWQ2oiH25aLO+QQ8MdpKzSoM0FMAaLfydDseSVoWLu3ceEKIEfj6MK/x1HCf8GoYCAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2df2680394ad6ccf35a575e05936aa270bab857dba126b18475ddf2ca2af8cc1","last_reissued_at":"2026-07-05T09:20:07.411161Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:20:07.411161Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ImagineNav: Prompting Vision-Language Models as Embodied Navigator through Scene Imagination","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Likun Tang, Teng Wang, Wenzhe Cai, Xinxin Zhao","submitted_at":"2024-10-13T15:31:31Z","abstract_excerpt":"Visual navigation is an essential skill for home-assistance robots, providing the object-searching ability to accomplish long-horizon daily tasks. Many recent approaches use Large Language Models (LLMs) for commonsense inference to improve exploration efficiency. However, the planning process of LLMs is limited within texts and it is difficult to represent the spatial occupancy and geometry layout only by texts. Both are important for making rational navigation decisions. In this work, we seek to unleash the spatial perception and planning ability of Vision-Language Models (VLMs), and explore "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.09874","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.09874/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.09874","created_at":"2026-07-05T09:20:07.411223+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.09874v1","created_at":"2026-07-05T09:20:07.411223+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.09874","created_at":"2026-07-05T09:20:07.411223+00:00"},{"alias_kind":"pith_short_12","alias_value":"FXZGQA4UVVWM","created_at":"2026-07-05T09:20:07.411223+00:00"},{"alias_kind":"pith_short_16","alias_value":"FXZGQA4UVVWM6NNF","created_at":"2026-07-05T09:20:07.411223+00:00"},{"alias_kind":"pith_short_8","alias_value":"FXZGQA4U","created_at":"2026-07-05T09:20:07.411223+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06165","citing_title":"EAGOR: Embodied Reasoning in Omni-direction","ref_index":28,"is_internal_anchor":true},{"citing_arxiv_id":"2606.31071","citing_title":"Hierarchical 3D Scene Graph Construction and Belief-based Planning for Semantic Navigation","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06223","citing_title":"ProCompNav: Proactive Instance Navigation with Comparative Judgment for Ambiguous User Queries","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06223","citing_title":"ProCompNav: Proactive Instance Navigation with Comparative Judgment for Ambiguous User Queries","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06223","citing_title":"ProCompNav: Proactive Instance Navigation with Comparative Judgment for Ambiguous User Queries","ref_index":39,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FXZGQA4UVVWM6NNFOXQFSNVKE4","json":"https://pith.science/pith/FXZGQA4UVVWM6NNFOXQFSNVKE4.json","graph_json":"https://pith.science/api/pith-number/FXZGQA4UVVWM6NNFOXQFSNVKE4/graph.json","events_json":"https://pith.science/api/pith-number/FXZGQA4UVVWM6NNFOXQFSNVKE4/events.json","paper":"https://pith.science/paper/FXZGQA4U"},"agent_actions":{"view_html":"https://pith.science/pith/FXZGQA4UVVWM6NNFOXQFSNVKE4","download_json":"https://pith.science/pith/FXZGQA4UVVWM6NNFOXQFSNVKE4.json","view_paper":"https://pith.science/paper/FXZGQA4U","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.09874&json=true","fetch_graph":"https://pith.science/api/pith-number/FXZGQA4UVVWM6NNFOXQFSNVKE4/graph.json","fetch_events":"https://pith.science/api/pith-number/FXZGQA4UVVWM6NNFOXQFSNVKE4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FXZGQA4UVVWM6NNFOXQFSNVKE4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FXZGQA4UVVWM6NNFOXQFSNVKE4/action/storage_attestation","attest_author":"https://pith.science/pith/FXZGQA4UVVWM6NNFOXQFSNVKE4/action/author_attestation","sign_citation":"https://pith.science/pith/FXZGQA4UVVWM6NNFOXQFSNVKE4/action/citation_signature","submit_replication":"https://pith.science/pith/FXZGQA4UVVWM6NNFOXQFSNVKE4/action/replication_record"}},"created_at":"2026-07-05T09:20:07.411223+00:00","updated_at":"2026-07-05T09:20:07.411223+00:00"}