{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:OEBMILWSLNG4FMOLI74Q7RYNPI","short_pith_number":"pith:OEBMILWS","schema_version":"1.0","canonical_sha256":"7102c42ed25b4dc2b1cb47f90fc70d7a0ead8d20e7e3b9f201d015d2739864bd","source":{"kind":"arxiv","id":"2109.13202","version":2},"attestation_state":"computed","paper":{"title":"MiniHack the Planet: A Sandbox for Open-Ended Reinforcement Learning Research","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Edward Grefenstette, Eric Hambro, Fabio Petroni, Heinrich K\\\"uttler, Jack Parker-Holder, Mikayel Samvelyan, Minqi Jiang, Robert Kirk, Tim Rockt\\\"aschel, Vitaly Kurin","submitted_at":"2021-09-27T17:22:42Z","abstract_excerpt":"Progress in deep reinforcement learning (RL) is heavily driven by the availability of challenging benchmarks used for training agents. However, benchmarks that are widely adopted by the community are not explicitly designed for evaluating specific capabilities of RL methods. While there exist environments for assessing particular open problems in RL (such as exploration, transfer learning, unsupervised environment design, or even language-assisted RL), it is generally difficult to extend these to richer, more complex environments once research goes beyond proof-of-concept results. We present M"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2109.13202","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-09-27T17:22:42Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"84bde3adf547f74fec634c1109ee106bd74e681cd9f2843ee95292c2a200cd31","abstract_canon_sha256":"bbae0fc7b7eefd4b09d1d818e0d76328220c3b43d4f2bc2850a31cf002c5338d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:32:38.996723Z","signature_b64":"r0gZxRHD5XVbxL7/pmVZIrKRRWRQG4Yr6AzYmaGKmDx+hV24cflddnCOs1gHpXiAz0wYvgRN+azy161pp7vxBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7102c42ed25b4dc2b1cb47f90fc70d7a0ead8d20e7e3b9f201d015d2739864bd","last_reissued_at":"2026-07-05T03:32:38.996187Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:32:38.996187Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MiniHack the Planet: A Sandbox for Open-Ended Reinforcement Learning Research","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Edward Grefenstette, Eric Hambro, Fabio Petroni, Heinrich K\\\"uttler, Jack Parker-Holder, Mikayel Samvelyan, Minqi Jiang, Robert Kirk, Tim Rockt\\\"aschel, Vitaly Kurin","submitted_at":"2021-09-27T17:22:42Z","abstract_excerpt":"Progress in deep reinforcement learning (RL) is heavily driven by the availability of challenging benchmarks used for training agents. However, benchmarks that are widely adopted by the community are not explicitly designed for evaluating specific capabilities of RL methods. While there exist environments for assessing particular open problems in RL (such as exploration, transfer learning, unsupervised environment design, or even language-assisted RL), it is generally difficult to extend these to richer, more complex environments once research goes beyond proof-of-concept results. We present M"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2109.13202","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2109.13202/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2109.13202","created_at":"2026-07-05T03:32:38.996248+00:00"},{"alias_kind":"arxiv_version","alias_value":"2109.13202v2","created_at":"2026-07-05T03:32:38.996248+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2109.13202","created_at":"2026-07-05T03:32:38.996248+00:00"},{"alias_kind":"pith_short_12","alias_value":"OEBMILWSLNG4","created_at":"2026-07-05T03:32:38.996248+00:00"},{"alias_kind":"pith_short_16","alias_value":"OEBMILWSLNG4FMOL","created_at":"2026-07-05T03:32:38.996248+00:00"},{"alias_kind":"pith_short_8","alias_value":"OEBMILWS","created_at":"2026-07-05T03:32:38.996248+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01224","citing_title":"AutoMem: Automated Learning of Memory as a Cognitive Skill","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2309.16797","citing_title":"Promptbreeder: Self-Referential Self-Improvement Via Prompt Evolution","ref_index":281,"is_internal_anchor":false},{"citing_arxiv_id":"2306.03310","citing_title":"LIBERO: Benchmarking Knowledge Transfer for Lifelong Robot Learning","ref_index":58,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OEBMILWSLNG4FMOLI74Q7RYNPI","json":"https://pith.science/pith/OEBMILWSLNG4FMOLI74Q7RYNPI.json","graph_json":"https://pith.science/api/pith-number/OEBMILWSLNG4FMOLI74Q7RYNPI/graph.json","events_json":"https://pith.science/api/pith-number/OEBMILWSLNG4FMOLI74Q7RYNPI/events.json","paper":"https://pith.science/paper/OEBMILWS"},"agent_actions":{"view_html":"https://pith.science/pith/OEBMILWSLNG4FMOLI74Q7RYNPI","download_json":"https://pith.science/pith/OEBMILWSLNG4FMOLI74Q7RYNPI.json","view_paper":"https://pith.science/paper/OEBMILWS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2109.13202&json=true","fetch_graph":"https://pith.science/api/pith-number/OEBMILWSLNG4FMOLI74Q7RYNPI/graph.json","fetch_events":"https://pith.science/api/pith-number/OEBMILWSLNG4FMOLI74Q7RYNPI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OEBMILWSLNG4FMOLI74Q7RYNPI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OEBMILWSLNG4FMOLI74Q7RYNPI/action/storage_attestation","attest_author":"https://pith.science/pith/OEBMILWSLNG4FMOLI74Q7RYNPI/action/author_attestation","sign_citation":"https://pith.science/pith/OEBMILWSLNG4FMOLI74Q7RYNPI/action/citation_signature","submit_replication":"https://pith.science/pith/OEBMILWSLNG4FMOLI74Q7RYNPI/action/replication_record"}},"created_at":"2026-07-05T03:32:38.996248+00:00","updated_at":"2026-07-05T03:32:38.996248+00:00"}