{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:HTGPYT3FUENYYJETUCIBRS6HVS","short_pith_number":"pith:HTGPYT3F","schema_version":"1.0","canonical_sha256":"3cccfc4f65a11b8c2493a09018cbc7ac826ad693459c278ffb197281fbe76c31","source":{"kind":"arxiv","id":"2105.06453","version":2},"attestation_state":"computed","paper":{"title":"Episodic Transformer for Vision-and-Language Navigation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Alexander Pashevich, Chen Sun, Cordelia Schmid","submitted_at":"2021-05-13T17:51:46Z","abstract_excerpt":"Interaction and navigation defined by natural language instructions in dynamic environments pose significant challenges for neural agents. This paper focuses on addressing two challenges: handling long sequence of subtasks, and understanding complex human instructions. We propose Episodic Transformer (E.T.), a multimodal transformer that encodes language inputs and the full episode history of visual observations and actions. To improve training, we leverage synthetic instructions as an intermediate representation that decouples understanding the visual appearance of an environment from the var"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2105.06453","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2021-05-13T17:51:46Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"49196a4c80f9aac27805a66ce80f2f6e3a7bfed51a1c60e467adfaf0074df3ac","abstract_canon_sha256":"1ff82e5446177ed5469bc6281cfe8f378cfc23a8ed4b81900349fe05140d386b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:08:45.729883Z","signature_b64":"NO3DSlD67s+P4vrw0XWu6dOjNPeJf3o2c6JIGpb/wrv3xHzHBn7PXIk8DHWHtF8Pz8qaGuAA/GFiZy0RjusmCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3cccfc4f65a11b8c2493a09018cbc7ac826ad693459c278ffb197281fbe76c31","last_reissued_at":"2026-07-05T03:08:45.729460Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:08:45.729460Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Episodic Transformer for Vision-and-Language Navigation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Alexander Pashevich, Chen Sun, Cordelia Schmid","submitted_at":"2021-05-13T17:51:46Z","abstract_excerpt":"Interaction and navigation defined by natural language instructions in dynamic environments pose significant challenges for neural agents. This paper focuses on addressing two challenges: handling long sequence of subtasks, and understanding complex human instructions. We propose Episodic Transformer (E.T.), a multimodal transformer that encodes language inputs and the full episode history of visual observations and actions. To improve training, we leverage synthetic instructions as an intermediate representation that decouples understanding the visual appearance of an environment from the var"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2105.06453","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2105.06453/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2105.06453","created_at":"2026-07-05T03:08:45.729523+00:00"},{"alias_kind":"arxiv_version","alias_value":"2105.06453v2","created_at":"2026-07-05T03:08:45.729523+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2105.06453","created_at":"2026-07-05T03:08:45.729523+00:00"},{"alias_kind":"pith_short_12","alias_value":"HTGPYT3FUENY","created_at":"2026-07-05T03:08:45.729523+00:00"},{"alias_kind":"pith_short_16","alias_value":"HTGPYT3FUENYYJET","created_at":"2026-07-05T03:08:45.729523+00:00"},{"alias_kind":"pith_short_8","alias_value":"HTGPYT3F","created_at":"2026-07-05T03:08:45.729523+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.13446","citing_title":"CAST: Counterfactual Labels Improve Instruction Following in Vision-Language-Action Models","ref_index":17,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HTGPYT3FUENYYJETUCIBRS6HVS","json":"https://pith.science/pith/HTGPYT3FUENYYJETUCIBRS6HVS.json","graph_json":"https://pith.science/api/pith-number/HTGPYT3FUENYYJETUCIBRS6HVS/graph.json","events_json":"https://pith.science/api/pith-number/HTGPYT3FUENYYJETUCIBRS6HVS/events.json","paper":"https://pith.science/paper/HTGPYT3F"},"agent_actions":{"view_html":"https://pith.science/pith/HTGPYT3FUENYYJETUCIBRS6HVS","download_json":"https://pith.science/pith/HTGPYT3FUENYYJETUCIBRS6HVS.json","view_paper":"https://pith.science/paper/HTGPYT3F","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2105.06453&json=true","fetch_graph":"https://pith.science/api/pith-number/HTGPYT3FUENYYJETUCIBRS6HVS/graph.json","fetch_events":"https://pith.science/api/pith-number/HTGPYT3FUENYYJETUCIBRS6HVS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HTGPYT3FUENYYJETUCIBRS6HVS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HTGPYT3FUENYYJETUCIBRS6HVS/action/storage_attestation","attest_author":"https://pith.science/pith/HTGPYT3FUENYYJETUCIBRS6HVS/action/author_attestation","sign_citation":"https://pith.science/pith/HTGPYT3FUENYYJETUCIBRS6HVS/action/citation_signature","submit_replication":"https://pith.science/pith/HTGPYT3FUENYYJETUCIBRS6HVS/action/replication_record"}},"created_at":"2026-07-05T03:08:45.729523+00:00","updated_at":"2026-07-05T03:08:45.729523+00:00"}