{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:PWQHQVKILJ3CLCV4S7PHANVS3M","short_pith_number":"pith:PWQHQVKI","schema_version":"1.0","canonical_sha256":"7da07855485a76258abc97de7036b2db11651f3c209ccd0f528a31361eabe16f","source":{"kind":"arxiv","id":"2110.02439","version":2},"attestation_state":"computed","paper":{"title":"Replay-Guided Adversarial Environment Design","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Edward Grefenstette, Jack Parker-Holder, Jakob Foerster, Michael Dennis, Minqi Jiang, Tim Rockt\\\"aschel","submitted_at":"2021-10-06T01:01:39Z","abstract_excerpt":"Deep reinforcement learning (RL) agents may successfully generalize to new settings if trained on an appropriately diverse set of environment and task configurations. Unsupervised Environment Design (UED) is a promising self-supervised RL paradigm, wherein the free parameters of an underspecified environment are automatically adapted during training to the agent's capabilities, leading to the emergence of diverse training environments. Here, we cast Prioritized Level Replay (PLR), an empirically successful but theoretically unmotivated method that selectively samples randomly-generated trainin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2110.02439","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-10-06T01:01:39Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"03df99a6a68c66a4e198f4bc752bf8a2c97cb1c4cdbf12c22036a431b73c53e8","abstract_canon_sha256":"22ac7d5501dae54b464fe74d89b0da56b3aa598561c8e9c8bedfb23d83a5beda"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:48:19.858058Z","signature_b64":"tcAaS74BEmQAfqq0LGVT6Izi/3P3OmsKIi8bTRTWOx2x6HStGQ//GDJyAQzpi+io91owfNGOW+4Mfa490ADTBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7da07855485a76258abc97de7036b2db11651f3c209ccd0f528a31361eabe16f","last_reissued_at":"2026-07-05T03:48:19.857596Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:48:19.857596Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Replay-Guided Adversarial Environment Design","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Edward Grefenstette, Jack Parker-Holder, Jakob Foerster, Michael Dennis, Minqi Jiang, Tim Rockt\\\"aschel","submitted_at":"2021-10-06T01:01:39Z","abstract_excerpt":"Deep reinforcement learning (RL) agents may successfully generalize to new settings if trained on an appropriately diverse set of environment and task configurations. Unsupervised Environment Design (UED) is a promising self-supervised RL paradigm, wherein the free parameters of an underspecified environment are automatically adapted during training to the agent's capabilities, leading to the emergence of diverse training environments. Here, we cast Prioritized Level Replay (PLR), an empirically successful but theoretically unmotivated method that selectively samples randomly-generated trainin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2110.02439","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2110.02439/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2110.02439","created_at":"2026-07-05T03:48:19.857660+00:00"},{"alias_kind":"arxiv_version","alias_value":"2110.02439v2","created_at":"2026-07-05T03:48:19.857660+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2110.02439","created_at":"2026-07-05T03:48:19.857660+00:00"},{"alias_kind":"pith_short_12","alias_value":"PWQHQVKILJ3C","created_at":"2026-07-05T03:48:19.857660+00:00"},{"alias_kind":"pith_short_16","alias_value":"PWQHQVKILJ3CLCV4","created_at":"2026-07-05T03:48:19.857660+00:00"},{"alias_kind":"pith_short_8","alias_value":"PWQHQVKI","created_at":"2026-07-05T03:48:19.857660+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PWQHQVKILJ3CLCV4S7PHANVS3M","json":"https://pith.science/pith/PWQHQVKILJ3CLCV4S7PHANVS3M.json","graph_json":"https://pith.science/api/pith-number/PWQHQVKILJ3CLCV4S7PHANVS3M/graph.json","events_json":"https://pith.science/api/pith-number/PWQHQVKILJ3CLCV4S7PHANVS3M/events.json","paper":"https://pith.science/paper/PWQHQVKI"},"agent_actions":{"view_html":"https://pith.science/pith/PWQHQVKILJ3CLCV4S7PHANVS3M","download_json":"https://pith.science/pith/PWQHQVKILJ3CLCV4S7PHANVS3M.json","view_paper":"https://pith.science/paper/PWQHQVKI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2110.02439&json=true","fetch_graph":"https://pith.science/api/pith-number/PWQHQVKILJ3CLCV4S7PHANVS3M/graph.json","fetch_events":"https://pith.science/api/pith-number/PWQHQVKILJ3CLCV4S7PHANVS3M/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PWQHQVKILJ3CLCV4S7PHANVS3M/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PWQHQVKILJ3CLCV4S7PHANVS3M/action/storage_attestation","attest_author":"https://pith.science/pith/PWQHQVKILJ3CLCV4S7PHANVS3M/action/author_attestation","sign_citation":"https://pith.science/pith/PWQHQVKILJ3CLCV4S7PHANVS3M/action/citation_signature","submit_replication":"https://pith.science/pith/PWQHQVKILJ3CLCV4S7PHANVS3M/action/replication_record"}},"created_at":"2026-07-05T03:48:19.857660+00:00","updated_at":"2026-07-05T03:48:19.857660+00:00"}