{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:XIN4SSIKRA6XC7ERNEVGZYZFBH","short_pith_number":"pith:XIN4SSIK","schema_version":"1.0","canonical_sha256":"ba1bc9490a883d717c91692a6ce32509fb0dd281e132cb43a7b337ded59f7d05","source":{"kind":"arxiv","id":"2603.27577","version":2},"attestation_state":"computed","paper":{"title":"Structured Observation Language for Efficient and Generalizable Vision-Language Navigation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.CV","authors_text":"Daojie Peng, Fulong Ma, Jun Ma","submitted_at":"2026-03-29T08:34:05Z","abstract_excerpt":"Vision-Language Navigation (VLN) requires an embodied agent to navigate complex environments by following natural language instructions, which typically demands tight fusion of visual and language modalities. Existing VLN methods often convert raw images into visual tokens or implicit features, requiring large-scale visual pre-training and suffering from poor generalization under environmental variations (e.g., lighting, texture). To address these issues, we propose SOL-Nav (Structured Observation Language for Navigation), a novel framework that translates egocentric visual observations into c"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2603.27577","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2026-03-29T08:34:05Z","cross_cats_sorted":["cs.RO"],"title_canon_sha256":"41574a437a7bc695fe8c1a5379cf770dafbcbf1ccb6003353e0c9f184aaf1653","abstract_canon_sha256":"7d01e58585aa0f0064c2f9b2fdbcdfda2e4e1c82e752b26baed934783bb43c78"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-28T01:22:32.427368Z","signature_b64":"83tR/Th8VNw/gS3Kaqni04RzaxJ1I4GQkP1okoZUoG0dLOZis2UK6c96zUaHoOQQQ4BSanMJdqeLppd/ArrlDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ba1bc9490a883d717c91692a6ce32509fb0dd281e132cb43a7b337ded59f7d05","last_reissued_at":"2026-07-28T01:22:32.426392Z","signature_status":"signed_v1","first_computed_at":"2026-07-28T01:22:32.426392Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Structured Observation Language for Efficient and Generalizable Vision-Language Navigation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.CV","authors_text":"Daojie Peng, Fulong Ma, Jun Ma","submitted_at":"2026-03-29T08:34:05Z","abstract_excerpt":"Vision-Language Navigation (VLN) requires an embodied agent to navigate complex environments by following natural language instructions, which typically demands tight fusion of visual and language modalities. Existing VLN methods often convert raw images into visual tokens or implicit features, requiring large-scale visual pre-training and suffering from poor generalization under environmental variations (e.g., lighting, texture). To address these issues, we propose SOL-Nav (Structured Observation Language for Navigation), a novel framework that translates egocentric visual observations into c"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2603.27577","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2603.27577/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2603.27577","created_at":"2026-07-28T01:22:32.426859+00:00"},{"alias_kind":"arxiv_version","alias_value":"2603.27577v2","created_at":"2026-07-28T01:22:32.426859+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2603.27577","created_at":"2026-07-28T01:22:32.426859+00:00"},{"alias_kind":"pith_short_12","alias_value":"XIN4SSIKRA6X","created_at":"2026-07-28T01:22:32.426859+00:00"},{"alias_kind":"pith_short_16","alias_value":"XIN4SSIKRA6XC7ER","created_at":"2026-07-28T01:22:32.426859+00:00"},{"alias_kind":"pith_short_8","alias_value":"XIN4SSIK","created_at":"2026-07-28T01:22:32.426859+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":3,"sample":[{"citing_arxiv_id":"2606.03188","citing_title":"GeoSem-WAM: Geometry- and Semantic-Aware World Action Models","ref_index":5,"is_internal_anchor":true},{"citing_arxiv_id":"2605.13548","citing_title":"AttenA+: Rectifying Action Inequality in Robotic Foundation Models","ref_index":3,"is_internal_anchor":true},{"citing_arxiv_id":"2605.13548","citing_title":"AttenA+: Rectifying Action Inequality in Robotic Foundation Models","ref_index":3,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XIN4SSIKRA6XC7ERNEVGZYZFBH","json":"https://pith.science/pith/XIN4SSIKRA6XC7ERNEVGZYZFBH.json","graph_json":"https://pith.science/api/pith-number/XIN4SSIKRA6XC7ERNEVGZYZFBH/graph.json","events_json":"https://pith.science/api/pith-number/XIN4SSIKRA6XC7ERNEVGZYZFBH/events.json","paper":"https://pith.science/paper/XIN4SSIK"},"agent_actions":{"view_html":"https://pith.science/pith/XIN4SSIKRA6XC7ERNEVGZYZFBH","download_json":"https://pith.science/pith/XIN4SSIKRA6XC7ERNEVGZYZFBH.json","view_paper":"https://pith.science/paper/XIN4SSIK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2603.27577&json=true","fetch_graph":"https://pith.science/api/pith-number/XIN4SSIKRA6XC7ERNEVGZYZFBH/graph.json","fetch_events":"https://pith.science/api/pith-number/XIN4SSIKRA6XC7ERNEVGZYZFBH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XIN4SSIKRA6XC7ERNEVGZYZFBH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XIN4SSIKRA6XC7ERNEVGZYZFBH/action/storage_attestation","attest_author":"https://pith.science/pith/XIN4SSIKRA6XC7ERNEVGZYZFBH/action/author_attestation","sign_citation":"https://pith.science/pith/XIN4SSIKRA6XC7ERNEVGZYZFBH/action/citation_signature","submit_replication":"https://pith.science/pith/XIN4SSIKRA6XC7ERNEVGZYZFBH/action/replication_record"}},"created_at":"2026-07-28T01:22:32.426859+00:00","updated_at":"2026-07-28T01:22:32.426859+00:00"}