{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:SYTT46IITP5BJTD7GA6DYFIUGV","short_pith_number":"pith:SYTT46II","schema_version":"1.0","canonical_sha256":"96273e79089bfa14cc7f303c3c1514356fc3328b440fdcf7abc88697d801e469","source":{"kind":"arxiv","id":"2006.08875","version":2},"attestation_state":"computed","paper":{"title":"Model-based Adversarial Meta-Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Garrett Thomas, Guangwen Yang, Tengyu Ma, Zichuan Lin","submitted_at":"2020-06-16T02:21:49Z","abstract_excerpt":"Meta-reinforcement learning (meta-RL) aims to learn from multiple training tasks the ability to adapt efficiently to unseen test tasks. Despite the success, existing meta-RL algorithms are known to be sensitive to the task distribution shift. When the test task distribution is different from the training task distribution, the performance may degrade significantly. To address this issue, this paper proposes Model-based Adversarial Meta-Reinforcement Learning (AdMRL), where we aim to minimize the worst-case sub-optimality gap -- the difference between the optimal return and the return that the "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2006.08875","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-06-16T02:21:49Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"f444bf3ae2de168e9f0c6ce0d69980c88c283e8b899acac0e64382c5b3c823cd","abstract_canon_sha256":"66b9eea0412a02d9ce91745d1252bfcb9400cfff7973871d0d8027be62a34436"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:18:52.669296Z","signature_b64":"iB1DUb4FZ0oOOUcr1yJ663sb9UAvLvDm5S5awPYYLdjPv2vLNzBfqMD4buen7KFi6q4RXpvGeG94w/xj8jYMDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"96273e79089bfa14cc7f303c3c1514356fc3328b440fdcf7abc88697d801e469","last_reissued_at":"2026-07-05T02:18:52.668830Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:18:52.668830Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Model-based Adversarial Meta-Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Garrett Thomas, Guangwen Yang, Tengyu Ma, Zichuan Lin","submitted_at":"2020-06-16T02:21:49Z","abstract_excerpt":"Meta-reinforcement learning (meta-RL) aims to learn from multiple training tasks the ability to adapt efficiently to unseen test tasks. Despite the success, existing meta-RL algorithms are known to be sensitive to the task distribution shift. When the test task distribution is different from the training task distribution, the performance may degrade significantly. To address this issue, this paper proposes Model-based Adversarial Meta-Reinforcement Learning (AdMRL), where we aim to minimize the worst-case sub-optimality gap -- the difference between the optimal return and the return that the "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2006.08875","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2006.08875/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2006.08875","created_at":"2026-07-05T02:18:52.668890+00:00"},{"alias_kind":"arxiv_version","alias_value":"2006.08875v2","created_at":"2026-07-05T02:18:52.668890+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2006.08875","created_at":"2026-07-05T02:18:52.668890+00:00"},{"alias_kind":"pith_short_12","alias_value":"SYTT46IITP5B","created_at":"2026-07-05T02:18:52.668890+00:00"},{"alias_kind":"pith_short_16","alias_value":"SYTT46IITP5BJTD7","created_at":"2026-07-05T02:18:52.668890+00:00"},{"alias_kind":"pith_short_8","alias_value":"SYTT46II","created_at":"2026-07-05T02:18:52.668890+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SYTT46IITP5BJTD7GA6DYFIUGV","json":"https://pith.science/pith/SYTT46IITP5BJTD7GA6DYFIUGV.json","graph_json":"https://pith.science/api/pith-number/SYTT46IITP5BJTD7GA6DYFIUGV/graph.json","events_json":"https://pith.science/api/pith-number/SYTT46IITP5BJTD7GA6DYFIUGV/events.json","paper":"https://pith.science/paper/SYTT46II"},"agent_actions":{"view_html":"https://pith.science/pith/SYTT46IITP5BJTD7GA6DYFIUGV","download_json":"https://pith.science/pith/SYTT46IITP5BJTD7GA6DYFIUGV.json","view_paper":"https://pith.science/paper/SYTT46II","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2006.08875&json=true","fetch_graph":"https://pith.science/api/pith-number/SYTT46IITP5BJTD7GA6DYFIUGV/graph.json","fetch_events":"https://pith.science/api/pith-number/SYTT46IITP5BJTD7GA6DYFIUGV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SYTT46IITP5BJTD7GA6DYFIUGV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SYTT46IITP5BJTD7GA6DYFIUGV/action/storage_attestation","attest_author":"https://pith.science/pith/SYTT46IITP5BJTD7GA6DYFIUGV/action/author_attestation","sign_citation":"https://pith.science/pith/SYTT46IITP5BJTD7GA6DYFIUGV/action/citation_signature","submit_replication":"https://pith.science/pith/SYTT46IITP5BJTD7GA6DYFIUGV/action/replication_record"}},"created_at":"2026-07-05T02:18:52.668890+00:00","updated_at":"2026-07-05T02:18:52.668890+00:00"}