{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:45S5JGN3L2FF7TMUC32EIGHBEC","short_pith_number":"pith:45S5JGN3","schema_version":"1.0","canonical_sha256":"e765d499bb5e8a5fcd9416f44418e120baec2eb29aa68cdec493574a513f39fd","source":{"kind":"arxiv","id":"2505.00743","version":1},"attestation_state":"computed","paper":{"title":"DOPE: Dual Object Perception-Enhancement Network for Vision-and-Language Navigation","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.CV","authors_text":"Dongsheng Yang, Yinfeng Yu","submitted_at":"2025-04-30T06:47:13Z","abstract_excerpt":"Vision-and-Language Navigation (VLN) is a challenging task where an agent must understand language instructions and navigate unfamiliar environments using visual cues. The agent must accurately locate the target based on visual information from the environment and complete tasks through interaction with the surroundings. Despite significant advancements in this field, two major limitations persist: (1) Many existing methods input complete language instructions directly into multi-layer Transformer networks without fully exploiting the detailed information within the instructions, thereby limit"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.00743","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CV","submitted_at":"2025-04-30T06:47:13Z","cross_cats_sorted":["cs.RO"],"title_canon_sha256":"9755e0a4fdda0c0106aad4595d0d30e0e559f22952a858ba01b97b6fa25ede9d","abstract_canon_sha256":"2460217ab8d4a01a6d78406fd7e89029cdcdd5ccc3f71e907621f04a5add118e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:57:38.381932Z","signature_b64":"asL881l+B0hCXUQsiVZD5AUG+J4Q+1NSIeSiKFgtvXtLM0ssQJm8zhfjmnHPQWpRealQFeL6c/nGZNU3FhsODQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e765d499bb5e8a5fcd9416f44418e120baec2eb29aa68cdec493574a513f39fd","last_reissued_at":"2026-07-05T10:57:38.381394Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:57:38.381394Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DOPE: Dual Object Perception-Enhancement Network for Vision-and-Language Navigation","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.CV","authors_text":"Dongsheng Yang, Yinfeng Yu","submitted_at":"2025-04-30T06:47:13Z","abstract_excerpt":"Vision-and-Language Navigation (VLN) is a challenging task where an agent must understand language instructions and navigate unfamiliar environments using visual cues. The agent must accurately locate the target based on visual information from the environment and complete tasks through interaction with the surroundings. Despite significant advancements in this field, two major limitations persist: (1) Many existing methods input complete language instructions directly into multi-layer Transformer networks without fully exploiting the detailed information within the instructions, thereby limit"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.00743","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.00743/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.00743","created_at":"2026-07-05T10:57:38.381472+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.00743v1","created_at":"2026-07-05T10:57:38.381472+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.00743","created_at":"2026-07-05T10:57:38.381472+00:00"},{"alias_kind":"pith_short_12","alias_value":"45S5JGN3L2FF","created_at":"2026-07-05T10:57:38.381472+00:00"},{"alias_kind":"pith_short_16","alias_value":"45S5JGN3L2FF7TMU","created_at":"2026-07-05T10:57:38.381472+00:00"},{"alias_kind":"pith_short_8","alias_value":"45S5JGN3","created_at":"2026-07-05T10:57:38.381472+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/45S5JGN3L2FF7TMUC32EIGHBEC","json":"https://pith.science/pith/45S5JGN3L2FF7TMUC32EIGHBEC.json","graph_json":"https://pith.science/api/pith-number/45S5JGN3L2FF7TMUC32EIGHBEC/graph.json","events_json":"https://pith.science/api/pith-number/45S5JGN3L2FF7TMUC32EIGHBEC/events.json","paper":"https://pith.science/paper/45S5JGN3"},"agent_actions":{"view_html":"https://pith.science/pith/45S5JGN3L2FF7TMUC32EIGHBEC","download_json":"https://pith.science/pith/45S5JGN3L2FF7TMUC32EIGHBEC.json","view_paper":"https://pith.science/paper/45S5JGN3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.00743&json=true","fetch_graph":"https://pith.science/api/pith-number/45S5JGN3L2FF7TMUC32EIGHBEC/graph.json","fetch_events":"https://pith.science/api/pith-number/45S5JGN3L2FF7TMUC32EIGHBEC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/45S5JGN3L2FF7TMUC32EIGHBEC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/45S5JGN3L2FF7TMUC32EIGHBEC/action/storage_attestation","attest_author":"https://pith.science/pith/45S5JGN3L2FF7TMUC32EIGHBEC/action/author_attestation","sign_citation":"https://pith.science/pith/45S5JGN3L2FF7TMUC32EIGHBEC/action/citation_signature","submit_replication":"https://pith.science/pith/45S5JGN3L2FF7TMUC32EIGHBEC/action/replication_record"}},"created_at":"2026-07-05T10:57:38.381472+00:00","updated_at":"2026-07-05T10:57:38.381472+00:00"}