{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:S4XAQ4VBUD3CSKSOZWFQOUSE5I","short_pith_number":"pith:S4XAQ4VB","schema_version":"1.0","canonical_sha256":"972e0872a1a0f6292a4ecd8b075244ea34458e5b6eccf41a2a7ae018cd5369c5","source":{"kind":"arxiv","id":"2306.13831","version":1},"attestation_state":"computed","paper":{"title":"Minigrid & Miniworld: Modular & Customizable Reinforcement Learning Environments for Goal-Oriented Tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Bolun Dai, Jordan Terry, Lucas Willems, Mark Towers, Maxime Chevalier-Boisvert, Pablo Samuel Castro, Rodrigo de Lazcano, Salem Lahlou, Suman Pal","submitted_at":"2023-06-24T01:16:07Z","abstract_excerpt":"We present the Minigrid and Miniworld libraries which provide a suite of goal-oriented 2D and 3D environments. The libraries were explicitly created with a minimalistic design paradigm to allow users to rapidly develop new environments for a wide range of research-specific needs. As a result, both have received widescale adoption by the RL community, facilitating research in a wide range of areas. In this paper, we outline the design philosophy, environment details, and their world generation API. We also showcase the additional capabilities brought by the unified API between Minigrid and Mini"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.13831","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-06-24T01:16:07Z","cross_cats_sorted":[],"title_canon_sha256":"10f313ee6dc452d7222612fc6f2b4aae3f907e543be83c42da72880f58724be4","abstract_canon_sha256":"5c268e139371d01caa2eda7c316015bda58c6dacde5b2ada10106f2863608b8e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:24:16.434629Z","signature_b64":"Sy+ebwx3a4YYHYiGkDzcGQx8JBZ2cupks0OgEsLX4EKnzNoAzzQ1dlVF8DLGuqnQVv4mmXjPsqEl6DYjBFQ4DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"972e0872a1a0f6292a4ecd8b075244ea34458e5b6eccf41a2a7ae018cd5369c5","last_reissued_at":"2026-07-05T06:24:16.434135Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:24:16.434135Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Minigrid & Miniworld: Modular & Customizable Reinforcement Learning Environments for Goal-Oriented Tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Bolun Dai, Jordan Terry, Lucas Willems, Mark Towers, Maxime Chevalier-Boisvert, Pablo Samuel Castro, Rodrigo de Lazcano, Salem Lahlou, Suman Pal","submitted_at":"2023-06-24T01:16:07Z","abstract_excerpt":"We present the Minigrid and Miniworld libraries which provide a suite of goal-oriented 2D and 3D environments. The libraries were explicitly created with a minimalistic design paradigm to allow users to rapidly develop new environments for a wide range of research-specific needs. As a result, both have received widescale adoption by the RL community, facilitating research in a wide range of areas. In this paper, we outline the design philosophy, environment details, and their world generation API. We also showcase the additional capabilities brought by the unified API between Minigrid and Mini"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.13831","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.13831/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.13831","created_at":"2026-07-05T06:24:16.434209+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.13831v1","created_at":"2026-07-05T06:24:16.434209+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.13831","created_at":"2026-07-05T06:24:16.434209+00:00"},{"alias_kind":"pith_short_12","alias_value":"S4XAQ4VBUD3C","created_at":"2026-07-05T06:24:16.434209+00:00"},{"alias_kind":"pith_short_16","alias_value":"S4XAQ4VBUD3CSKSO","created_at":"2026-07-05T06:24:16.434209+00:00"},{"alias_kind":"pith_short_8","alias_value":"S4XAQ4VB","created_at":"2026-07-05T06:24:16.434209+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":19,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07753","citing_title":"A Transdiagnostic Space of Disorder Like Phenotypes in Reinforcement Learning Agents","ref_index":20,"is_internal_anchor":true},{"citing_arxiv_id":"2606.24622","citing_title":"Themis: An explainable AI-enabled framework for Reinforcement Learning with Human Feedback","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21350","citing_title":"A Reward-Petri-Net Interpretation of Temporal Behavior Trees","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02087","citing_title":"SUNTA: Hierarchical Video Prediction with Surprise-based Chunking","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02440","citing_title":"EvoPolicyGym: Evaluating Autonomous Policy Evolution in Interactive Environments","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12191","citing_title":"Agentic Environment Engineering for Large Language Models: A Survey of Environment Modeling, Synthesis, Evaluation, and Application","ref_index":104,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06673","citing_title":"Uncertainty-Aware LLM-Guided Policy Shaping for Sparse-Reward Reinforcement Learning","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26357","citing_title":"Balancing Plasticity and Stability with Fast and Slow Successor Features","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31444","citing_title":"Answer-Set-Programming-based Abstractions for Reinforcement Learning","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23372","citing_title":"Curriculum reinforcement learning with measurable task representation learning","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2402.05284","citing_title":"Analyzing Adversarial Inputs in Deep Reinforcement Learning","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2509.25424","citing_title":"Polychromic Objectives for Reinforcement Learning","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2509.25438","citing_title":"Beyond Noisy-TVs: Noise-Robust Exploration Via Learning Progress Monitoring","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2601.02850","citing_title":"Sample-Efficient Neurosymbolic Deep Reinforcement Learning","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12261","citing_title":"Delay-Empowered Causal Hierarchical Reinforcement Learning","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27162","citing_title":"A High-Throughput Compute-Efficient POMDP Hide-And-Seek-Engine (HASE) for Multi-Agent Operations","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25534","citing_title":"Sample-efficient Neuro-symbolic Proximal Policy Optimization","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14440","citing_title":"On Tackling Complex Tasks with Reward Machines and Signal Temporal Logics","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21640","citing_title":"Task-specific Subnetwork Discovery in Reinforcement Learning for Autonomous Underwater Navigation","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/S4XAQ4VBUD3CSKSOZWFQOUSE5I","json":"https://pith.science/pith/S4XAQ4VBUD3CSKSOZWFQOUSE5I.json","graph_json":"https://pith.science/api/pith-number/S4XAQ4VBUD3CSKSOZWFQOUSE5I/graph.json","events_json":"https://pith.science/api/pith-number/S4XAQ4VBUD3CSKSOZWFQOUSE5I/events.json","paper":"https://pith.science/paper/S4XAQ4VB"},"agent_actions":{"view_html":"https://pith.science/pith/S4XAQ4VBUD3CSKSOZWFQOUSE5I","download_json":"https://pith.science/pith/S4XAQ4VBUD3CSKSOZWFQOUSE5I.json","view_paper":"https://pith.science/paper/S4XAQ4VB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.13831&json=true","fetch_graph":"https://pith.science/api/pith-number/S4XAQ4VBUD3CSKSOZWFQOUSE5I/graph.json","fetch_events":"https://pith.science/api/pith-number/S4XAQ4VBUD3CSKSOZWFQOUSE5I/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/S4XAQ4VBUD3CSKSOZWFQOUSE5I/action/timestamp_anchor","attest_storage":"https://pith.science/pith/S4XAQ4VBUD3CSKSOZWFQOUSE5I/action/storage_attestation","attest_author":"https://pith.science/pith/S4XAQ4VBUD3CSKSOZWFQOUSE5I/action/author_attestation","sign_citation":"https://pith.science/pith/S4XAQ4VBUD3CSKSOZWFQOUSE5I/action/citation_signature","submit_replication":"https://pith.science/pith/S4XAQ4VBUD3CSKSOZWFQOUSE5I/action/replication_record"}},"created_at":"2026-07-05T06:24:16.434209+00:00","updated_at":"2026-07-05T06:24:16.434209+00:00"}