{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:GYIVOERJSFD6XIZMMHWAXVXNUH","short_pith_number":"pith:GYIVOERJ","schema_version":"1.0","canonical_sha256":"36115712299147eba32c61ec0bd6eda1feda70ad1f124fb3e47dee4553c81a32","source":{"kind":"arxiv","id":"2007.04309","version":3},"attestation_state":"computed","paper":{"title":"Self-Supervised Policy Adaptation during Deployment","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","cs.RO","stat.ML"],"primary_cat":"cs.LG","authors_text":"Alexei A. Efros, Guillem Aleny\\`a, Lerrel Pinto, Nicklas Hansen, Pieter Abbeel, Rishabh Jangir, Xiaolong Wang, Yu Sun","submitted_at":"2020-07-08T17:56:27Z","abstract_excerpt":"In most real world scenarios, a policy trained by reinforcement learning in one environment needs to be deployed in another, potentially quite different environment. However, generalization across different environments is known to be hard. A natural solution would be to keep training after deployment in the new environment, but this cannot be done if the new environment offers no reward signal. Our work explores the use of self-supervision to allow the policy to continue training after deployment without using any rewards. While previous methods explicitly anticipate changes in the new enviro"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2007.04309","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-07-08T17:56:27Z","cross_cats_sorted":["cs.CV","cs.RO","stat.ML"],"title_canon_sha256":"d91c359abf2e2166093c033b9680981094735fa9cd4e90cfeee063e132d2ee93","abstract_canon_sha256":"8b17931a697d96f76cf592710b203e7d06c09692fa6028165901d5684b9f90bf"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:30:29.110870Z","signature_b64":"RsvO9KQtfI+4pDvogMQkH8kBbONktCsXSxSpq/hp5L2N1CzC9c0uagFh1OpKcM/6O0rfDIWQUL31wFkhmJD9Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"36115712299147eba32c61ec0bd6eda1feda70ad1f124fb3e47dee4553c81a32","last_reissued_at":"2026-07-05T02:30:29.110370Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:30:29.110370Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Self-Supervised Policy Adaptation during Deployment","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","cs.RO","stat.ML"],"primary_cat":"cs.LG","authors_text":"Alexei A. Efros, Guillem Aleny\\`a, Lerrel Pinto, Nicklas Hansen, Pieter Abbeel, Rishabh Jangir, Xiaolong Wang, Yu Sun","submitted_at":"2020-07-08T17:56:27Z","abstract_excerpt":"In most real world scenarios, a policy trained by reinforcement learning in one environment needs to be deployed in another, potentially quite different environment. However, generalization across different environments is known to be hard. A natural solution would be to keep training after deployment in the new environment, but this cannot be done if the new environment offers no reward signal. Our work explores the use of self-supervision to allow the policy to continue training after deployment without using any rewards. While previous methods explicitly anticipate changes in the new enviro"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2007.04309","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2007.04309/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2007.04309","created_at":"2026-07-05T02:30:29.110430+00:00"},{"alias_kind":"arxiv_version","alias_value":"2007.04309v3","created_at":"2026-07-05T02:30:29.110430+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2007.04309","created_at":"2026-07-05T02:30:29.110430+00:00"},{"alias_kind":"pith_short_12","alias_value":"GYIVOERJSFD6","created_at":"2026-07-05T02:30:29.110430+00:00"},{"alias_kind":"pith_short_16","alias_value":"GYIVOERJSFD6XIZM","created_at":"2026-07-05T02:30:29.110430+00:00"},{"alias_kind":"pith_short_8","alias_value":"GYIVOERJ","created_at":"2026-07-05T02:30:29.110430+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26515","citing_title":"Forget, Anticipate and Adapt: Test Time Training for Long Videos","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03127","citing_title":"TTT-VLA: Test-Time Latent Prompt Optimization for Vision-Language-Action Models","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26515","citing_title":"Forget, Anticipate and Adapt: Test Time Training for Long Videos","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30192","citing_title":"Domain Adaptation with Adaptive Imagination for Visual Reinforcement Learning under Limited Target Data","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2507.13662","citing_title":"Iteratively Learning Muscle Memory for Legged Robots to Master Adaptive and High Precision Locomotion","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2302.11550","citing_title":"Scaling Robot Learning with Semantically Imagined Experience","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2407.04620","citing_title":"Learning to (Learn at Test Time): RNNs with Expressive Hidden States","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05857","citing_title":"Offline Reinforcement Learning for Rotation Profile Control in Tokamaks","ref_index":49,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GYIVOERJSFD6XIZMMHWAXVXNUH","json":"https://pith.science/pith/GYIVOERJSFD6XIZMMHWAXVXNUH.json","graph_json":"https://pith.science/api/pith-number/GYIVOERJSFD6XIZMMHWAXVXNUH/graph.json","events_json":"https://pith.science/api/pith-number/GYIVOERJSFD6XIZMMHWAXVXNUH/events.json","paper":"https://pith.science/paper/GYIVOERJ"},"agent_actions":{"view_html":"https://pith.science/pith/GYIVOERJSFD6XIZMMHWAXVXNUH","download_json":"https://pith.science/pith/GYIVOERJSFD6XIZMMHWAXVXNUH.json","view_paper":"https://pith.science/paper/GYIVOERJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2007.04309&json=true","fetch_graph":"https://pith.science/api/pith-number/GYIVOERJSFD6XIZMMHWAXVXNUH/graph.json","fetch_events":"https://pith.science/api/pith-number/GYIVOERJSFD6XIZMMHWAXVXNUH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GYIVOERJSFD6XIZMMHWAXVXNUH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GYIVOERJSFD6XIZMMHWAXVXNUH/action/storage_attestation","attest_author":"https://pith.science/pith/GYIVOERJSFD6XIZMMHWAXVXNUH/action/author_attestation","sign_citation":"https://pith.science/pith/GYIVOERJSFD6XIZMMHWAXVXNUH/action/citation_signature","submit_replication":"https://pith.science/pith/GYIVOERJSFD6XIZMMHWAXVXNUH/action/replication_record"}},"created_at":"2026-07-05T02:30:29.110430+00:00","updated_at":"2026-07-05T02:30:29.110430+00:00"}