{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:FNJF4RWHKCZJRMYDHU4WUZGAOJ","short_pith_number":"pith:FNJF4RWH","schema_version":"1.0","canonical_sha256":"2b525e46c750b298b3033d396a64c0724c154300c3db8c600b63476027c96e26","source":{"kind":"arxiv","id":"2507.13152","version":3},"attestation_state":"computed","paper":{"title":"SE-VLN: A Self-Evolving Vision-Language Navigation Framework Based on Multimodal Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.CV","authors_text":"Fuhai Chen, Haoran Zhao, Haozhou Li, Jiang Gao, Juan Liu, Xiangyu Dong, Xiaoguang Ma, Yaoming Zhou","submitted_at":"2025-07-17T14:13:50Z","abstract_excerpt":"Recent advances in vision-language navigation (VLN) were mainly attributed to emerging large language models (LLMs). These methods exhibited excellent generalization capabilities in instruction understanding and task reasoning. However, they were constrained by the fixed knowledge bases and reasoning abilities of LLMs, preventing fully incorporating experiential knowledge and thus resulting in a lack of efficient evolutionary capacity. To address this, we drew inspiration from the evolution capabilities of natural agents, and proposed a self-evolving VLN framework (SE-VLN) to endow VLN agents "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.13152","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-07-17T14:13:50Z","cross_cats_sorted":["cs.AI","cs.RO"],"title_canon_sha256":"4860377c3cf66438c442f1a2b3cef46d8f49bc9ed45c8b86cecb0e9ee88fd052","abstract_canon_sha256":"f6a30358b63e848afa8ab1358bcf5fe489e18ba4aa8ac35bd180633d1de030fe"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:59:17.978919Z","signature_b64":"octY2i5joskC6m8Pw8OZWCZMkXkt483KICrmLOXroo9G8g4pLVEa/I5iQV5WMu5VSI7Dw76t0qdzMMA0+Bm+Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2b525e46c750b298b3033d396a64c0724c154300c3db8c600b63476027c96e26","last_reissued_at":"2026-07-05T11:59:17.978409Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:59:17.978409Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SE-VLN: A Self-Evolving Vision-Language Navigation Framework Based on Multimodal Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.CV","authors_text":"Fuhai Chen, Haoran Zhao, Haozhou Li, Jiang Gao, Juan Liu, Xiangyu Dong, Xiaoguang Ma, Yaoming Zhou","submitted_at":"2025-07-17T14:13:50Z","abstract_excerpt":"Recent advances in vision-language navigation (VLN) were mainly attributed to emerging large language models (LLMs). These methods exhibited excellent generalization capabilities in instruction understanding and task reasoning. However, they were constrained by the fixed knowledge bases and reasoning abilities of LLMs, preventing fully incorporating experiential knowledge and thus resulting in a lack of efficient evolutionary capacity. To address this, we drew inspiration from the evolution capabilities of natural agents, and proposed a self-evolving VLN framework (SE-VLN) to endow VLN agents "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.13152","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.13152/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.13152","created_at":"2026-07-05T11:59:17.978472+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.13152v3","created_at":"2026-07-05T11:59:17.978472+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.13152","created_at":"2026-07-05T11:59:17.978472+00:00"},{"alias_kind":"pith_short_12","alias_value":"FNJF4RWHKCZJ","created_at":"2026-07-05T11:59:17.978472+00:00"},{"alias_kind":"pith_short_16","alias_value":"FNJF4RWHKCZJRMYD","created_at":"2026-07-05T11:59:17.978472+00:00"},{"alias_kind":"pith_short_8","alias_value":"FNJF4RWH","created_at":"2026-07-05T11:59:17.978472+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01043","citing_title":"DART-VLN: Test-Time Memory Decay and Anti-Loop Regularization for Discrete Vision-Language Navigation","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28397","citing_title":"CLOSER-VLN: Closed-Loop Self-Verified Retrieval-Augmented Reasoning for Aerial Vision-Language Navigation","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10118","citing_title":"Plan in Sandbox, Navigate in Open Worlds: Learning Physics-Grounded Abstracted Experience for Embodied Navigation","ref_index":51,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FNJF4RWHKCZJRMYDHU4WUZGAOJ","json":"https://pith.science/pith/FNJF4RWHKCZJRMYDHU4WUZGAOJ.json","graph_json":"https://pith.science/api/pith-number/FNJF4RWHKCZJRMYDHU4WUZGAOJ/graph.json","events_json":"https://pith.science/api/pith-number/FNJF4RWHKCZJRMYDHU4WUZGAOJ/events.json","paper":"https://pith.science/paper/FNJF4RWH"},"agent_actions":{"view_html":"https://pith.science/pith/FNJF4RWHKCZJRMYDHU4WUZGAOJ","download_json":"https://pith.science/pith/FNJF4RWHKCZJRMYDHU4WUZGAOJ.json","view_paper":"https://pith.science/paper/FNJF4RWH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.13152&json=true","fetch_graph":"https://pith.science/api/pith-number/FNJF4RWHKCZJRMYDHU4WUZGAOJ/graph.json","fetch_events":"https://pith.science/api/pith-number/FNJF4RWHKCZJRMYDHU4WUZGAOJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FNJF4RWHKCZJRMYDHU4WUZGAOJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FNJF4RWHKCZJRMYDHU4WUZGAOJ/action/storage_attestation","attest_author":"https://pith.science/pith/FNJF4RWHKCZJRMYDHU4WUZGAOJ/action/author_attestation","sign_citation":"https://pith.science/pith/FNJF4RWHKCZJRMYDHU4WUZGAOJ/action/citation_signature","submit_replication":"https://pith.science/pith/FNJF4RWHKCZJRMYDHU4WUZGAOJ/action/replication_record"}},"created_at":"2026-07-05T11:59:17.978472+00:00","updated_at":"2026-07-05T11:59:17.978472+00:00"}