{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:2HWVGOT5J7MNYQHOCDPLCZYZRW","short_pith_number":"pith:2HWVGOT5","schema_version":"1.0","canonical_sha256":"d1ed533a7d4fd8dc40ee10deb167198d92c40a8210cd7d70dcb393c226612404","source":{"kind":"arxiv","id":"2402.10670","version":2},"attestation_state":"computed","paper":{"title":"OpenFMNav: Towards Open-Set Zero-Shot Object Navigation via Vision-Language Foundation Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.CL","authors_text":"Hai Lin, Meng Jiang, Yuxuan Kuang","submitted_at":"2024-02-16T13:21:33Z","abstract_excerpt":"Object navigation (ObjectNav) requires an agent to navigate through unseen environments to find queried objects. Many previous methods attempted to solve this task by relying on supervised or reinforcement learning, where they are trained on limited household datasets with close-set objects. However, two key challenges are unsolved: understanding free-form natural language instructions that demand open-set objects, and generalizing to new environments in a zero-shot manner. Aiming to solve the two challenges, in this paper, we propose OpenFMNav, an Open-set Foundation Model based framework for"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.10670","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-02-16T13:21:33Z","cross_cats_sorted":["cs.RO"],"title_canon_sha256":"47ec6d2d9c0693183f5de11ce80fc12196d3c3a96bba865c92e717eaad38d5f2","abstract_canon_sha256":"fe139b38159421899aee9ddc7364456b7edab02f8bd6c95c5dac85f04d1a885b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:00:01.434201Z","signature_b64":"lidv48ffLPo4uWWhSU0hijj+CtmQ2hSWwXbNPDcPLoXCrHtXAnI/wVKOg39UvrDl1age4l07ffIFO/m4C+1nDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d1ed533a7d4fd8dc40ee10deb167198d92c40a8210cd7d70dcb393c226612404","last_reissued_at":"2026-07-05T08:00:01.433820Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:00:01.433820Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"OpenFMNav: Towards Open-Set Zero-Shot Object Navigation via Vision-Language Foundation Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.CL","authors_text":"Hai Lin, Meng Jiang, Yuxuan Kuang","submitted_at":"2024-02-16T13:21:33Z","abstract_excerpt":"Object navigation (ObjectNav) requires an agent to navigate through unseen environments to find queried objects. Many previous methods attempted to solve this task by relying on supervised or reinforcement learning, where they are trained on limited household datasets with close-set objects. However, two key challenges are unsolved: understanding free-form natural language instructions that demand open-set objects, and generalizing to new environments in a zero-shot manner. Aiming to solve the two challenges, in this paper, we propose OpenFMNav, an Open-set Foundation Model based framework for"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.10670","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.10670/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.10670","created_at":"2026-07-05T08:00:01.433875+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.10670v2","created_at":"2026-07-05T08:00:01.433875+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.10670","created_at":"2026-07-05T08:00:01.433875+00:00"},{"alias_kind":"pith_short_12","alias_value":"2HWVGOT5J7MN","created_at":"2026-07-05T08:00:01.433875+00:00"},{"alias_kind":"pith_short_16","alias_value":"2HWVGOT5J7MNYQHO","created_at":"2026-07-05T08:00:01.433875+00:00"},{"alias_kind":"pith_short_8","alias_value":"2HWVGOT5","created_at":"2026-07-05T08:00:01.433875+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":16,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18112","citing_title":"Qwen-RobotNav Technical Report: A Scalable Navigation Model Designed for an Agentic Navigation System","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18235","citing_title":"EvolveNav: Proactive Preflection and Self-Evolving Memory for Zero-Shot Object Goal Navigation","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10927","citing_title":"AllDayNav: Lifelong Navigation via Real-World Reinforcement Learning","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31919","citing_title":"MVP-Nav: Multi-layer Value Map Planner Navigator","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31071","citing_title":"Hierarchical 3D Scene Graph Construction and Belief-based Planning for Semantic Navigation","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18112","citing_title":"Qwen-RobotNav Technical Report: A Scalable Navigation Model Designed for an Agentic Navigation System","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27582","citing_title":"Uni-LaViRA: Language-Vision-Robot Actions Translation for Unified Embodied Navigation","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19206","citing_title":"CLUE: Adaptively Prioritized Contextual Cues by Leveraging a Unified Semantic Map for Effective Zero-Shot Object-Goal Navigation","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19594","citing_title":"MCNav: Memory-Aware Dynamic Cognitive Map for Zero-shot Goal-oriented Navigation","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16979","citing_title":"NORM-Nav: Zero-Shot Mobile Robot Navigation with Natural Language Behavioral Constraints","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2402.15852","citing_title":"NaVid: Video-based VLM Plans the Next Step for Vision-and-Language Navigation","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2510.20685","citing_title":"C-NAV: Towards Self-Evolving Continual Object Navigation in Open World","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2412.06224","citing_title":"Uni-NaVid: A Video-based Vision-Language-Action Model for Unifying Embodied Navigation Tasks","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03139","citing_title":"FSUNav: A Cerebrum-Cerebellum Architecture for Fast, Safe, and Universal Zero-Shot Goal-Oriented Navigation","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05960","citing_title":"Plug-and-Play Label Map Diffusion for Universal Goal-Oriented Navigation","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12872","citing_title":"OVAL: Open-Vocabulary Augmented Memory Model for Lifelong Object Goal Navigation","ref_index":42,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2HWVGOT5J7MNYQHOCDPLCZYZRW","json":"https://pith.science/pith/2HWVGOT5J7MNYQHOCDPLCZYZRW.json","graph_json":"https://pith.science/api/pith-number/2HWVGOT5J7MNYQHOCDPLCZYZRW/graph.json","events_json":"https://pith.science/api/pith-number/2HWVGOT5J7MNYQHOCDPLCZYZRW/events.json","paper":"https://pith.science/paper/2HWVGOT5"},"agent_actions":{"view_html":"https://pith.science/pith/2HWVGOT5J7MNYQHOCDPLCZYZRW","download_json":"https://pith.science/pith/2HWVGOT5J7MNYQHOCDPLCZYZRW.json","view_paper":"https://pith.science/paper/2HWVGOT5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.10670&json=true","fetch_graph":"https://pith.science/api/pith-number/2HWVGOT5J7MNYQHOCDPLCZYZRW/graph.json","fetch_events":"https://pith.science/api/pith-number/2HWVGOT5J7MNYQHOCDPLCZYZRW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2HWVGOT5J7MNYQHOCDPLCZYZRW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2HWVGOT5J7MNYQHOCDPLCZYZRW/action/storage_attestation","attest_author":"https://pith.science/pith/2HWVGOT5J7MNYQHOCDPLCZYZRW/action/author_attestation","sign_citation":"https://pith.science/pith/2HWVGOT5J7MNYQHOCDPLCZYZRW/action/citation_signature","submit_replication":"https://pith.science/pith/2HWVGOT5J7MNYQHOCDPLCZYZRW/action/replication_record"}},"created_at":"2026-07-05T08:00:01.433875+00:00","updated_at":"2026-07-05T08:00:01.433875+00:00"}