{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:IDK6ZEQBZAUQYQ3N4AZ2MLZXEB","short_pith_number":"pith:IDK6ZEQB","schema_version":"1.0","canonical_sha256":"40d5ec9201c8290c436de033a62f37205e8f38fa72859b630564e7f1c189431c","source":{"kind":"arxiv","id":"2104.05632","version":3},"attestation_state":"computed","paper":{"title":"Augmented World Models Facilitate Zero-Shot Dynamics Generalization From a Single Offline Environment","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Cong Lu, Jack Parker-Holder, Philip J. Ball, Stephen Roberts","submitted_at":"2021-04-12T16:53:55Z","abstract_excerpt":"Reinforcement learning from large-scale offline datasets provides us with the ability to learn policies without potentially unsafe or impractical exploration. Significant progress has been made in the past few years in dealing with the challenge of correcting for differing behavior between the data collection and learned policies. However, little attention has been paid to potentially changing dynamics when transferring a policy to the online setting, where performance can be up to 90% reduced for existing methods. In this paper we address this problem with Augmented World Models (AugWM). We a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2104.05632","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-04-12T16:53:55Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"2bbbe771adee8b6c33291d547ad6c0ea55d1de7d1737038ee19145cd04205813","abstract_canon_sha256":"bec1877c44a75138d95e5cb23a5e4a6a92da0982a15174f3e74b8849f507d419"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:02:50.518967Z","signature_b64":"oiccwmwYrwPdgGHaoiHrWpzZttsF8tpwOfqpsrsaB/oWz+Cqh9+m+4in3hG5fOtL2nbZglf2zuOI9bGfUz93CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"40d5ec9201c8290c436de033a62f37205e8f38fa72859b630564e7f1c189431c","last_reissued_at":"2026-07-05T03:02:50.518491Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:02:50.518491Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Augmented World Models Facilitate Zero-Shot Dynamics Generalization From a Single Offline Environment","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Cong Lu, Jack Parker-Holder, Philip J. Ball, Stephen Roberts","submitted_at":"2021-04-12T16:53:55Z","abstract_excerpt":"Reinforcement learning from large-scale offline datasets provides us with the ability to learn policies without potentially unsafe or impractical exploration. Significant progress has been made in the past few years in dealing with the challenge of correcting for differing behavior between the data collection and learned policies. However, little attention has been paid to potentially changing dynamics when transferring a policy to the online setting, where performance can be up to 90% reduced for existing methods. In this paper we address this problem with Augmented World Models (AugWM). We a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2104.05632","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2104.05632/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2104.05632","created_at":"2026-07-05T03:02:50.518552+00:00"},{"alias_kind":"arxiv_version","alias_value":"2104.05632v3","created_at":"2026-07-05T03:02:50.518552+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2104.05632","created_at":"2026-07-05T03:02:50.518552+00:00"},{"alias_kind":"pith_short_12","alias_value":"IDK6ZEQBZAUQ","created_at":"2026-07-05T03:02:50.518552+00:00"},{"alias_kind":"pith_short_16","alias_value":"IDK6ZEQBZAUQYQ3N","created_at":"2026-07-05T03:02:50.518552+00:00"},{"alias_kind":"pith_short_8","alias_value":"IDK6ZEQB","created_at":"2026-07-05T03:02:50.518552+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IDK6ZEQBZAUQYQ3N4AZ2MLZXEB","json":"https://pith.science/pith/IDK6ZEQBZAUQYQ3N4AZ2MLZXEB.json","graph_json":"https://pith.science/api/pith-number/IDK6ZEQBZAUQYQ3N4AZ2MLZXEB/graph.json","events_json":"https://pith.science/api/pith-number/IDK6ZEQBZAUQYQ3N4AZ2MLZXEB/events.json","paper":"https://pith.science/paper/IDK6ZEQB"},"agent_actions":{"view_html":"https://pith.science/pith/IDK6ZEQBZAUQYQ3N4AZ2MLZXEB","download_json":"https://pith.science/pith/IDK6ZEQBZAUQYQ3N4AZ2MLZXEB.json","view_paper":"https://pith.science/paper/IDK6ZEQB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2104.05632&json=true","fetch_graph":"https://pith.science/api/pith-number/IDK6ZEQBZAUQYQ3N4AZ2MLZXEB/graph.json","fetch_events":"https://pith.science/api/pith-number/IDK6ZEQBZAUQYQ3N4AZ2MLZXEB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IDK6ZEQBZAUQYQ3N4AZ2MLZXEB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IDK6ZEQBZAUQYQ3N4AZ2MLZXEB/action/storage_attestation","attest_author":"https://pith.science/pith/IDK6ZEQBZAUQYQ3N4AZ2MLZXEB/action/author_attestation","sign_citation":"https://pith.science/pith/IDK6ZEQBZAUQYQ3N4AZ2MLZXEB/action/citation_signature","submit_replication":"https://pith.science/pith/IDK6ZEQBZAUQYQ3N4AZ2MLZXEB/action/replication_record"}},"created_at":"2026-07-05T03:02:50.518552+00:00","updated_at":"2026-07-05T03:02:50.518552+00:00"}