{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2020:ORVRNYM5EGV32WE76EZMLW5QXL","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"41de4bf522bee6255f1bd3f6bf42dc8747c8fbcefe93b0fe7528b9e04e2a3bd0","cross_cats_sorted":["stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-06-12T13:34:46Z","title_canon_sha256":"e34e474e2895022c89dcea8a4e01560eb8e561510b5c05e88e118030f61f9c02"},"schema_version":"1.0","source":{"id":"2006.07178","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2006.07178","created_at":"2026-07-05T01:10:40Z"},{"alias_kind":"arxiv_version","alias_value":"2006.07178v2","created_at":"2026-07-05T01:10:40Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2006.07178","created_at":"2026-07-05T01:10:40Z"},{"alias_kind":"pith_short_12","alias_value":"ORVRNYM5EGV3","created_at":"2026-07-05T01:10:40Z"},{"alias_kind":"pith_short_16","alias_value":"ORVRNYM5EGV32WE7","created_at":"2026-07-05T01:10:40Z"},{"alias_kind":"pith_short_8","alias_value":"ORVRNYM5","created_at":"2026-07-05T01:10:40Z"}],"graph_snapshots":[{"event_id":"sha256:b8a1744f1084059942238f1ccac9848488588ff124965f4b0c8a52a5c3eff681","target":"graph","created_at":"2026-07-05T01:10:40Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2006.07178/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement learning algorithms can acquire policies for complex tasks autonomously. However, the number of samples required to learn a diverse set of skills can be prohibitively large. While meta-reinforcement learning methods have enabled agents to leverage prior experience to adapt quickly to new tasks, their performance depends crucially on how close the new task is to the previously experienced tasks. Current approaches are either not able to extrapolate well, or can do so at the expense of requiring extremely large amounts of data for on-policy meta-training. In this work, we present m","authors_text":"Chelsea Finn, Russell Mendonca, Sergey Levine, Xinyang Geng","cross_cats":["stat.ML"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-06-12T13:34:46Z","title":"Meta-Reinforcement Learning Robust to Distributional Shift via Model Identification and Experience Relabeling"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2006.07178","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:a65b5d85cd91af6f5c8022a4ac9b71e8c60632bca764cc02f6f650cac8b8b4d6","target":"record","created_at":"2026-07-05T01:10:40Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"41de4bf522bee6255f1bd3f6bf42dc8747c8fbcefe93b0fe7528b9e04e2a3bd0","cross_cats_sorted":["stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-06-12T13:34:46Z","title_canon_sha256":"e34e474e2895022c89dcea8a4e01560eb8e561510b5c05e88e118030f61f9c02"},"schema_version":"1.0","source":{"id":"2006.07178","kind":"arxiv","version":2}},"canonical_sha256":"746b16e19d21abbd589ff132c5dbb0bae97c848da415a618264ca8e13fe6446f","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"746b16e19d21abbd589ff132c5dbb0bae97c848da415a618264ca8e13fe6446f","first_computed_at":"2026-07-05T01:10:40.955150Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T01:10:40.955150Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"e+3aeTIqNG15RRuQfuq4KXclr2z1XxYUVPdCGnlqh9ByWrmgj1wVHldX2fWMGFGljO67ssRpAKxWZN1YXdokBA==","signature_status":"signed_v1","signed_at":"2026-07-05T01:10:40.955648Z","signed_message":"canonical_sha256_bytes"},"source_id":"2006.07178","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:a65b5d85cd91af6f5c8022a4ac9b71e8c60632bca764cc02f6f650cac8b8b4d6","sha256:b8a1744f1084059942238f1ccac9848488588ff124965f4b0c8a52a5c3eff681"],"state_sha256":"d47aad30fc0aa1d6a2c718bd4cd1b84c117bd3b9ac0696c5f15c443dc356d911"}