{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:XY7KON7NUNNCW425TSTDLG4KXI","short_pith_number":"pith:XY7KON7N","schema_version":"1.0","canonical_sha256":"be3ea737eda35a2b735d9ca6359b8aba32c26507b6581b404d4350cfa690fcbf","source":{"kind":"arxiv","id":"2111.09794","version":6},"attestation_state":"computed","paper":{"title":"A Survey of Zero-shot Generalisation in Deep Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Amy Zhang, Edward Grefenstette, Robert Kirk, Tim Rockt\\\"aschel","submitted_at":"2021-11-18T16:53:02Z","abstract_excerpt":"The study of zero-shot generalisation (ZSG) in deep Reinforcement Learning (RL) aims to produce RL algorithms whose policies generalise well to novel unseen situations at deployment time, avoiding overfitting to their training environments. Tackling this is vital if we are to deploy reinforcement learning algorithms in real world scenarios, where the environment will be diverse, dynamic and unpredictable. This survey is an overview of this nascent field. We rely on a unifying formalism and terminology for discussing different ZSG problems, building upon previous works. We go on to categorise e"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2111.09794","kind":"arxiv","version":6},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-11-18T16:53:02Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"443dc01b529d399abe2cbe72f19303300ca011189e7c80b40235eed258ed7cb0","abstract_canon_sha256":"af2933a9f10806b0fe5cd67c052da33043ed830f90f678b6c6485b9d9d775c11"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:34:14.188818Z","signature_b64":"nawQMv1t6q3Bx9atwhQdngxVl1e6+JpkRCxxpteN62bPdNNeH7+zIUvQnDHRa5446R6HB1q9xgQSIPbhtqknAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"be3ea737eda35a2b735d9ca6359b8aba32c26507b6581b404d4350cfa690fcbf","last_reissued_at":"2026-07-05T05:34:14.188303Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:34:14.188303Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Survey of Zero-shot Generalisation in Deep Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Amy Zhang, Edward Grefenstette, Robert Kirk, Tim Rockt\\\"aschel","submitted_at":"2021-11-18T16:53:02Z","abstract_excerpt":"The study of zero-shot generalisation (ZSG) in deep Reinforcement Learning (RL) aims to produce RL algorithms whose policies generalise well to novel unseen situations at deployment time, avoiding overfitting to their training environments. Tackling this is vital if we are to deploy reinforcement learning algorithms in real world scenarios, where the environment will be diverse, dynamic and unpredictable. This survey is an overview of this nascent field. We rely on a unifying formalism and terminology for discussing different ZSG problems, building upon previous works. We go on to categorise e"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2111.09794","kind":"arxiv","version":6},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2111.09794/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2111.09794","created_at":"2026-07-05T05:34:14.188386+00:00"},{"alias_kind":"arxiv_version","alias_value":"2111.09794v6","created_at":"2026-07-05T05:34:14.188386+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2111.09794","created_at":"2026-07-05T05:34:14.188386+00:00"},{"alias_kind":"pith_short_12","alias_value":"XY7KON7NUNNC","created_at":"2026-07-05T05:34:14.188386+00:00"},{"alias_kind":"pith_short_16","alias_value":"XY7KON7NUNNCW425","created_at":"2026-07-05T05:34:14.188386+00:00"},{"alias_kind":"pith_short_8","alias_value":"XY7KON7N","created_at":"2026-07-05T05:34:14.188386+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03201","citing_title":"Reinforcement Learning from Cross-domain Videos with Video Prediction Model","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14201","citing_title":"MAPLE: Latent Multi-Agent Play for End-to-End Autonomous Driving","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2309.16797","citing_title":"Promptbreeder: Self-Referential Self-Improvement Via Prompt Evolution","ref_index":180,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14201","citing_title":"MAPLE: Latent Multi-Agent Play for End-to-End Autonomous Driving","ref_index":25,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XY7KON7NUNNCW425TSTDLG4KXI","json":"https://pith.science/pith/XY7KON7NUNNCW425TSTDLG4KXI.json","graph_json":"https://pith.science/api/pith-number/XY7KON7NUNNCW425TSTDLG4KXI/graph.json","events_json":"https://pith.science/api/pith-number/XY7KON7NUNNCW425TSTDLG4KXI/events.json","paper":"https://pith.science/paper/XY7KON7N"},"agent_actions":{"view_html":"https://pith.science/pith/XY7KON7NUNNCW425TSTDLG4KXI","download_json":"https://pith.science/pith/XY7KON7NUNNCW425TSTDLG4KXI.json","view_paper":"https://pith.science/paper/XY7KON7N","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2111.09794&json=true","fetch_graph":"https://pith.science/api/pith-number/XY7KON7NUNNCW425TSTDLG4KXI/graph.json","fetch_events":"https://pith.science/api/pith-number/XY7KON7NUNNCW425TSTDLG4KXI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XY7KON7NUNNCW425TSTDLG4KXI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XY7KON7NUNNCW425TSTDLG4KXI/action/storage_attestation","attest_author":"https://pith.science/pith/XY7KON7NUNNCW425TSTDLG4KXI/action/author_attestation","sign_citation":"https://pith.science/pith/XY7KON7NUNNCW425TSTDLG4KXI/action/citation_signature","submit_replication":"https://pith.science/pith/XY7KON7NUNNCW425TSTDLG4KXI/action/replication_record"}},"created_at":"2026-07-05T05:34:14.188386+00:00","updated_at":"2026-07-05T05:34:14.188386+00:00"}