{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2018:IUSU7BX236XGQO6S2XA3DPA465","short_pith_number":"pith:IUSU7BX2","schema_version":"1.0","canonical_sha256":"45254f86fadfae683bd2d5c1b1bc1cf74a89529d144fca0c213d25f8fa922efd","source":{"kind":"arxiv","id":"1801.00690","version":1},"attestation_state":"computed","paper":{"title":"DeepMind Control Suite","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"The DeepMind Control Suite offers a standardized set of continuous control tasks to benchmark reinforcement learning agents.","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Abbas Abdolmaleki, Alistair Muldal, Andrew Lefrancq, David Budden, Diego de las Casas, Josh Merel, Martin Riedmiller, Timothy Lillicrap, Tom Erez, Yazhe Li, Yotam Doron, Yuval Tassa","submitted_at":"2018-01-02T15:48:14Z","abstract_excerpt":"The DeepMind Control Suite is a set of continuous control tasks with a standardised structure and interpretable rewards, intended to serve as performance benchmarks for reinforcement learning agents. The tasks are written in Python and powered by the MuJoCo physics engine, making them easy to use and modify. We include benchmarks for several learning algorithms. The Control Suite is publicly available at https://www.github.com/deepmind/dm_control . A video summary of all tasks is available at http://youtu.be/rAai4QzcYbs ."},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":true,"formal_links_present":true},"canonical_record":{"source":{"id":"1801.00690","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2018-01-02T15:48:14Z","cross_cats_sorted":[],"title_canon_sha256":"11aa661c4e00a25af435dfbf79cbf609c7eb3068d3616fb0872f8cc672192052","abstract_canon_sha256":"09ca9a5aa7251c16ad0513268c06ebd7e89db5ac2fd395f5670995b8bfb1c5ae"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-04T22:27:08.426053Z","signature_b64":"yce4Z/bvxGblX88pHPcXSvbuL2Dz9FDU8FgBCTok5n6OGvJEf4KYkDEOALk1oBbAvUs8t6UYDtfh+LF4paCJBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"45254f86fadfae683bd2d5c1b1bc1cf74a89529d144fca0c213d25f8fa922efd","last_reissued_at":"2026-07-04T22:27:08.425426Z","signature_status":"signed_v1","first_computed_at":"2026-07-04T22:27:08.425426Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DeepMind Control Suite","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"The DeepMind Control Suite offers a standardized set of continuous control tasks to benchmark reinforcement learning agents.","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Abbas Abdolmaleki, Alistair Muldal, Andrew Lefrancq, David Budden, Diego de las Casas, Josh Merel, Martin Riedmiller, Timothy Lillicrap, Tom Erez, Yazhe Li, Yotam Doron, Yuval Tassa","submitted_at":"2018-01-02T15:48:14Z","abstract_excerpt":"The DeepMind Control Suite is a set of continuous control tasks with a standardised structure and interpretable rewards, intended to serve as performance benchmarks for reinforcement learning agents. The tasks are written in Python and powered by the MuJoCo physics engine, making them easy to use and modify. We include benchmarks for several learning algorithms. The Control Suite is publicly available at https://www.github.com/deepmind/dm_control . A video summary of all tasks is available at http://youtu.be/rAai4QzcYbs ."},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"The DeepMind Control Suite is a set of continuous control tasks with a standardised structure and interpretable rewards, intended to serve as performance benchmarks for reinforcement learning agents.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"That the chosen tasks and reward functions are sufficiently representative of real-world control problems and that performance on them will generalize to other domains.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"The DeepMind Control Suite supplies a standardized collection of continuous control tasks with interpretable rewards for benchmarking reinforcement learning agents.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"The DeepMind Control Suite offers a standardized set of continuous control tasks to benchmark reinforcement learning agents.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"35e364546e00c43de8e47c783972fb0ab04c430091fffc7d8091df1e5ab9431c"},"source":{"id":"1801.00690","kind":"arxiv","version":1},"verdict":{"id":"61aa29e9-f274-4932-b57f-82ac0b44b4f4","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-13T07:40:16.913388Z","strongest_claim":"The DeepMind Control Suite is a set of continuous control tasks with a standardised structure and interpretable rewards, intended to serve as performance benchmarks for reinforcement learning agents.","one_line_summary":"The DeepMind Control Suite supplies a standardized collection of continuous control tasks with interpretable rewards for benchmarking reinforcement learning agents.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"That the chosen tasks and reward functions are sufficiently representative of real-world control problems and that performance on them will generalize to other domains.","pith_extraction_headline":"The DeepMind Control Suite offers a standardized set of continuous control tasks to benchmark reinforcement learning agents."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1801.00690/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":13,"sample":[{"doi":"","year":null,"title":"Layer Normalization","work_id":"20a2d720-0046-4c7c-bcd6-327ec8143f69","ref_index":1,"cited_arxiv_id":"1607.06450","is_internal_anchor":true},{"doi":"10.1109/tsmc.1983.6313077","year":1983,"title":"doi: 10.1109/TSMC.1983.6313077","work_id":"5d5d0663-101c-45b7-a8b9-acfc8151271a","ref_index":2,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":null,"title":"A Distributional Perspective on Reinforcement Learning","work_id":"79879eb4-5aa1-4764-9aaa-505a0a1c0f7f","ref_index":3,"cited_arxiv_id":"1707.06887","is_internal_anchor":false},{"doi":"","year":2015,"title":"Simulation tools for model-based robotics: Comparison of bullet, havok, mujoco, ode and physx","work_id":"38cf2474-efac-4a24-9dac-7447fb98b1c4","ref_index":4,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":null,"title":"Reproducibility of Benchmarked Deep Reinforcement Learning Tasks for Continuous Control","work_id":"79bc5b68-7f90-4779-aa0d-5e000bdf6421","ref_index":5,"cited_arxiv_id":"1708.04133","is_internal_anchor":false}],"resolved_work":13,"snapshot_sha256":"e5d3d9256b696963fdf555bcc4d70a5a914b595253ea8fb696d4e8fac5a9c342","internal_anchors":4},"formal_canon":{"evidence_count":1,"snapshot_sha256":"9d2968cd73d6b711ea9d4f0230d56d8b47099a9c81078f92f6d2894550110cf7"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1801.00690","created_at":"2026-07-04T22:27:08.425485+00:00"},{"alias_kind":"arxiv_version","alias_value":"1801.00690v1","created_at":"2026-07-04T22:27:08.425485+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1801.00690","created_at":"2026-07-04T22:27:08.425485+00:00"},{"alias_kind":"pith_short_12","alias_value":"IUSU7BX236XG","created_at":"2026-07-04T22:27:08.425485+00:00"},{"alias_kind":"pith_short_16","alias_value":"IUSU7BX236XGQO6S","created_at":"2026-07-04T22:27:08.425485+00:00"},{"alias_kind":"pith_short_8","alias_value":"IUSU7BX2","created_at":"2026-07-04T22:27:08.425485+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":92,"internal_anchor_count":92,"sample":[{"citing_arxiv_id":"2607.06401","citing_title":"A Definition and Roadmap for World Models","ref_index":232,"is_internal_anchor":true},{"citing_arxiv_id":"2606.26392","citing_title":"MPC-Injection: Biasing Off-Policy Locomotion RL Toward Controller-Induced Behavior Basins","ref_index":48,"is_internal_anchor":true},{"citing_arxiv_id":"2606.26217","citing_title":"Fast LeWorldModel","ref_index":34,"is_internal_anchor":true},{"citing_arxiv_id":"2606.26574","citing_title":"Revisiting Action Factorization for Complex Action Spaces","ref_index":34,"is_internal_anchor":true},{"citing_arxiv_id":"2606.21086","citing_title":"ReFPO: Reflow Regularization for Flow Matching Policy Gradients","ref_index":33,"is_internal_anchor":true},{"citing_arxiv_id":"2606.21297","citing_title":"NASDAQ: Normalized Observation Space Dynamics-Augmented Q-Learning","ref_index":34,"is_internal_anchor":true},{"citing_arxiv_id":"2606.20104","citing_title":"Sensorimotor World Models: Perception for Action via Inverse Dynamics","ref_index":52,"is_internal_anchor":true},{"citing_arxiv_id":"2606.20376","citing_title":"CRAX: Fast Safe Reinforcement Learning Benchmarking","ref_index":38,"is_internal_anchor":true},{"citing_arxiv_id":"2606.18697","citing_title":"Stealthy World Model Manipulation via Data Poisoning","ref_index":7,"is_internal_anchor":true},{"citing_arxiv_id":"2607.02403","citing_title":"ACID: Action Consistency via Inverse Dynamics for Planning with World Models","ref_index":39,"is_internal_anchor":true},{"citing_arxiv_id":"2606.27962","citing_title":"Building a Scalable, Reproducible, Evaluatable, and Closed-Loop Simulation Environment Foundation for Embodied Intelligence","ref_index":35,"is_internal_anchor":true},{"citing_arxiv_id":"2607.00392","citing_title":"Learning Generalizable Skill Policy with Data-Efficient Unsupervised RL","ref_index":38,"is_internal_anchor":true},{"citing_arxiv_id":"2607.00811","citing_title":"From Pixels to Temporal Correlations: Learning Informative Representations for Reinforcement Learning Pre-training","ref_index":31,"is_internal_anchor":true},{"citing_arxiv_id":"2607.00796","citing_title":"Task-Relevant Representation Decoupling for Visual Reinforcement Learning Generalization","ref_index":32,"is_internal_anchor":true},{"citing_arxiv_id":"2607.00634","citing_title":"Loss Smoothing for Stable Adaptation Under Distribution Shift","ref_index":7,"is_internal_anchor":true},{"citing_arxiv_id":"2607.00642","citing_title":"Coachable agents for interactive gameplay","ref_index":54,"is_internal_anchor":true},{"citing_arxiv_id":"2606.06746","citing_title":"Performance Variation in Deep Reinforcement Learning","ref_index":7,"is_internal_anchor":true},{"citing_arxiv_id":"2606.05555","citing_title":"Representation Learning Enables Scalable Multitask Deep Reinforcement Learning","ref_index":146,"is_internal_anchor":true},{"citing_arxiv_id":"2606.03382","citing_title":"Local Guidance, Global Impact: Gaussian-Reshaped Trust Region Unlocks Behavior Transitions","ref_index":13,"is_internal_anchor":true},{"citing_arxiv_id":"2606.03521","citing_title":"Post-Hoc Robustness for Model-Based Reinforcement Learning","ref_index":50,"is_internal_anchor":true},{"citing_arxiv_id":"2606.01123","citing_title":"From Reward-Free Representations to Preferences: Rethinking Offline Preference-Based Reinforcement Learning","ref_index":135,"is_internal_anchor":true},{"citing_arxiv_id":"2606.27962","citing_title":"Building a Scalable, Reproducible, Evaluatable, and Closed-Loop Simulation Environment Foundation for Embodied Intelligence","ref_index":35,"is_internal_anchor":true},{"citing_arxiv_id":"2605.26379","citing_title":"When Does LeJEPA Learn a World Model?","ref_index":83,"is_internal_anchor":true},{"citing_arxiv_id":"2606.31232","citing_title":"Delta-JEPA: Learning Action-Sensitive World Models via Latent Difference Decoding","ref_index":12,"is_internal_anchor":true},{"citing_arxiv_id":"2605.23993","citing_title":"Nano World Models: A Minimalist Implementation of Future Video Prediction","ref_index":25,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":1,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IUSU7BX236XGQO6S2XA3DPA465","json":"https://pith.science/pith/IUSU7BX236XGQO6S2XA3DPA465.json","graph_json":"https://pith.science/api/pith-number/IUSU7BX236XGQO6S2XA3DPA465/graph.json","events_json":"https://pith.science/api/pith-number/IUSU7BX236XGQO6S2XA3DPA465/events.json","paper":"https://pith.science/paper/IUSU7BX2"},"agent_actions":{"view_html":"https://pith.science/pith/IUSU7BX236XGQO6S2XA3DPA465","download_json":"https://pith.science/pith/IUSU7BX236XGQO6S2XA3DPA465.json","view_paper":"https://pith.science/paper/IUSU7BX2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1801.00690&json=true","fetch_graph":"https://pith.science/api/pith-number/IUSU7BX236XGQO6S2XA3DPA465/graph.json","fetch_events":"https://pith.science/api/pith-number/IUSU7BX236XGQO6S2XA3DPA465/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IUSU7BX236XGQO6S2XA3DPA465/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IUSU7BX236XGQO6S2XA3DPA465/action/storage_attestation","attest_author":"https://pith.science/pith/IUSU7BX236XGQO6S2XA3DPA465/action/author_attestation","sign_citation":"https://pith.science/pith/IUSU7BX236XGQO6S2XA3DPA465/action/citation_signature","submit_replication":"https://pith.science/pith/IUSU7BX236XGQO6S2XA3DPA465/action/replication_record"}},"created_at":"2026-07-04T22:27:08.425485+00:00","updated_at":"2026-07-04T22:27:08.425485+00:00"}