{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:OWCFPIWIF7TKTDQISACQA427LB","short_pith_number":"pith:OWCFPIWI","schema_version":"1.0","canonical_sha256":"758457a2c82fe6a98e08900500735f585e886c1994bbc315f121611ce0ba044b","source":{"kind":"arxiv","id":"2007.10835","version":1},"attestation_state":"computed","paper":{"title":"Soft Expert Reward Learning for Vision-and-Language Navigation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chunhua Shen, Hu Wang, Qi Wu","submitted_at":"2020-07-21T14:17:36Z","abstract_excerpt":"Vision-and-Language Navigation (VLN) requires an agent to find a specified spot in an unseen environment by following natural language instructions. Dominant methods based on supervised learning clone expert's behaviours and thus perform better on seen environments, while showing restricted performance on unseen ones. Reinforcement Learning (RL) based models show better generalisation ability but have issues as well, requiring large amount of manual reward engineering is one of which. In this paper, we introduce a Soft Expert Reward Learning (SERL) model to overcome the reward engineering desi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2007.10835","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2020-07-21T14:17:36Z","cross_cats_sorted":[],"title_canon_sha256":"8e280d33d90bf9c00d5939a326f1f4c2b392f6f964eb837dc73fd07555a029dc","abstract_canon_sha256":"e43cbf9d21ec9d2d9f1f9cdf9846d2aa38254a0da6b97c4df923dd151562fba9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:21:12.484738Z","signature_b64":"XvfBInv8GLLyJ1I6sH9fxjDJXYKtzI3FCkFesjW6xDNjZKNWcw/NjyUr9a+NikmZp14KZRquzSGlCVTiQcyyAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"758457a2c82fe6a98e08900500735f585e886c1994bbc315f121611ce0ba044b","last_reissued_at":"2026-07-05T01:21:12.484297Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:21:12.484297Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Soft Expert Reward Learning for Vision-and-Language Navigation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chunhua Shen, Hu Wang, Qi Wu","submitted_at":"2020-07-21T14:17:36Z","abstract_excerpt":"Vision-and-Language Navigation (VLN) requires an agent to find a specified spot in an unseen environment by following natural language instructions. Dominant methods based on supervised learning clone expert's behaviours and thus perform better on seen environments, while showing restricted performance on unseen ones. Reinforcement Learning (RL) based models show better generalisation ability but have issues as well, requiring large amount of manual reward engineering is one of which. In this paper, we introduce a Soft Expert Reward Learning (SERL) model to overcome the reward engineering desi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2007.10835","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2007.10835/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2007.10835","created_at":"2026-07-05T01:21:12.484366+00:00"},{"alias_kind":"arxiv_version","alias_value":"2007.10835v1","created_at":"2026-07-05T01:21:12.484366+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2007.10835","created_at":"2026-07-05T01:21:12.484366+00:00"},{"alias_kind":"pith_short_12","alias_value":"OWCFPIWIF7TK","created_at":"2026-07-05T01:21:12.484366+00:00"},{"alias_kind":"pith_short_16","alias_value":"OWCFPIWIF7TKTDQI","created_at":"2026-07-05T01:21:12.484366+00:00"},{"alias_kind":"pith_short_8","alias_value":"OWCFPIWI","created_at":"2026-07-05T01:21:12.484366+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OWCFPIWIF7TKTDQISACQA427LB","json":"https://pith.science/pith/OWCFPIWIF7TKTDQISACQA427LB.json","graph_json":"https://pith.science/api/pith-number/OWCFPIWIF7TKTDQISACQA427LB/graph.json","events_json":"https://pith.science/api/pith-number/OWCFPIWIF7TKTDQISACQA427LB/events.json","paper":"https://pith.science/paper/OWCFPIWI"},"agent_actions":{"view_html":"https://pith.science/pith/OWCFPIWIF7TKTDQISACQA427LB","download_json":"https://pith.science/pith/OWCFPIWIF7TKTDQISACQA427LB.json","view_paper":"https://pith.science/paper/OWCFPIWI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2007.10835&json=true","fetch_graph":"https://pith.science/api/pith-number/OWCFPIWIF7TKTDQISACQA427LB/graph.json","fetch_events":"https://pith.science/api/pith-number/OWCFPIWIF7TKTDQISACQA427LB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OWCFPIWIF7TKTDQISACQA427LB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OWCFPIWIF7TKTDQISACQA427LB/action/storage_attestation","attest_author":"https://pith.science/pith/OWCFPIWIF7TKTDQISACQA427LB/action/author_attestation","sign_citation":"https://pith.science/pith/OWCFPIWIF7TKTDQISACQA427LB/action/citation_signature","submit_replication":"https://pith.science/pith/OWCFPIWIF7TKTDQISACQA427LB/action/replication_record"}},"created_at":"2026-07-05T01:21:12.484366+00:00","updated_at":"2026-07-05T01:21:12.484366+00:00"}