{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:TS2PYK7TM7N76TSV5GJEKELYNB","short_pith_number":"pith:TS2PYK7T","schema_version":"1.0","canonical_sha256":"9cb4fc2bf367dbff4e55e992451178685c0d42b1cc528da61b582b642c7b2b63","source":{"kind":"arxiv","id":"1912.01588","version":2},"attestation_state":"computed","paper":{"title":"Leveraging Procedural Generation to Benchmark Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Christopher Hesse, Jacob Hilton, John Schulman, Karl Cobbe","submitted_at":"2019-12-03T18:34:03Z","abstract_excerpt":"We introduce Procgen Benchmark, a suite of 16 procedurally generated game-like environments designed to benchmark both sample efficiency and generalization in reinforcement learning. We believe that the community will benefit from increased access to high quality training environments, and we provide detailed experimental protocols for using this benchmark. We empirically demonstrate that diverse environment distributions are essential to adequately train and evaluate RL agents, thereby motivating the extensive use of procedural content generation. We then use this benchmark to investigate the"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1912.01588","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-12-03T18:34:03Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"9a5fab83e2ee9e8af843363f8ff5a604154a6c0a9dcc8eaab62fa3059028c040","abstract_canon_sha256":"306d2164187ff9e37ee58ca4bb7068f375c87616f862f0bd4cf8df82ca0fdaa9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:22:15.210640Z","signature_b64":"aHEu1/7qUSeF3+3gbHF8l41ZLF+AHibPfYtYGZIKwXS+SlvNxBTVdDstn/78zW5XcVDQ0ZQYYOrNv3w7w1DuBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9cb4fc2bf367dbff4e55e992451178685c0d42b1cc528da61b582b642c7b2b63","last_reissued_at":"2026-07-05T01:22:15.210178Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:22:15.210178Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Leveraging Procedural Generation to Benchmark Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Christopher Hesse, Jacob Hilton, John Schulman, Karl Cobbe","submitted_at":"2019-12-03T18:34:03Z","abstract_excerpt":"We introduce Procgen Benchmark, a suite of 16 procedurally generated game-like environments designed to benchmark both sample efficiency and generalization in reinforcement learning. We believe that the community will benefit from increased access to high quality training environments, and we provide detailed experimental protocols for using this benchmark. We empirically demonstrate that diverse environment distributions are essential to adequately train and evaluate RL agents, thereby motivating the extensive use of procedural content generation. We then use this benchmark to investigate the"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1912.01588","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1912.01588/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1912.01588","created_at":"2026-07-05T01:22:15.210234+00:00"},{"alias_kind":"arxiv_version","alias_value":"1912.01588v2","created_at":"2026-07-05T01:22:15.210234+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1912.01588","created_at":"2026-07-05T01:22:15.210234+00:00"},{"alias_kind":"pith_short_12","alias_value":"TS2PYK7TM7N7","created_at":"2026-07-05T01:22:15.210234+00:00"},{"alias_kind":"pith_short_16","alias_value":"TS2PYK7TM7N76TSV","created_at":"2026-07-05T01:22:15.210234+00:00"},{"alias_kind":"pith_short_8","alias_value":"TS2PYK7T","created_at":"2026-07-05T01:22:15.210234+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18812","citing_title":"Reinforcement Learning Foundation Models Should Already Be A Thing","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04130","citing_title":"CLAW: Learning Continuous Latent Action World Models via Adversarial Latent Regularization","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23565","citing_title":"Understanding Goal Generalisation in Sequential Reinforcement Learning","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2407.15134","citing_title":"Proximal Policy Distillation","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2507.05561","citing_title":"Preemptive Solving of Future Problems: Multitask Preplay in Humans and Machines","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2309.16797","citing_title":"Promptbreeder: Self-Referential Self-Improvement Via Prompt Evolution","ref_index":83,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TS2PYK7TM7N76TSV5GJEKELYNB","json":"https://pith.science/pith/TS2PYK7TM7N76TSV5GJEKELYNB.json","graph_json":"https://pith.science/api/pith-number/TS2PYK7TM7N76TSV5GJEKELYNB/graph.json","events_json":"https://pith.science/api/pith-number/TS2PYK7TM7N76TSV5GJEKELYNB/events.json","paper":"https://pith.science/paper/TS2PYK7T"},"agent_actions":{"view_html":"https://pith.science/pith/TS2PYK7TM7N76TSV5GJEKELYNB","download_json":"https://pith.science/pith/TS2PYK7TM7N76TSV5GJEKELYNB.json","view_paper":"https://pith.science/paper/TS2PYK7T","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1912.01588&json=true","fetch_graph":"https://pith.science/api/pith-number/TS2PYK7TM7N76TSV5GJEKELYNB/graph.json","fetch_events":"https://pith.science/api/pith-number/TS2PYK7TM7N76TSV5GJEKELYNB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TS2PYK7TM7N76TSV5GJEKELYNB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TS2PYK7TM7N76TSV5GJEKELYNB/action/storage_attestation","attest_author":"https://pith.science/pith/TS2PYK7TM7N76TSV5GJEKELYNB/action/author_attestation","sign_citation":"https://pith.science/pith/TS2PYK7TM7N76TSV5GJEKELYNB/action/citation_signature","submit_replication":"https://pith.science/pith/TS2PYK7TM7N76TSV5GJEKELYNB/action/replication_record"}},"created_at":"2026-07-05T01:22:15.210234+00:00","updated_at":"2026-07-05T01:22:15.210234+00:00"}