{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:X4JKJ3ZE2TVGI7777B7IOTMG6B","short_pith_number":"pith:X4JKJ3ZE","schema_version":"1.0","canonical_sha256":"bf12a4ef24d4ea647ffff87e874d86f05856c7d9509c1a3506fe71d9cbb0750a","source":{"kind":"arxiv","id":"2208.07860","version":1},"attestation_state":"computed","paper":{"title":"A Walk in the Park: Learning to Walk in 20 Minutes With Model-Free Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.RO","authors_text":"Ilya Kostrikov, Laura Smith, Sergey Levine","submitted_at":"2022-08-16T17:37:36Z","abstract_excerpt":"Deep reinforcement learning is a promising approach to learning policies in uncontrolled environments that do not require domain knowledge. Unfortunately, due to sample inefficiency, deep RL applications have primarily focused on simulated environments. In this work, we demonstrate that the recent advancements in machine learning algorithms and libraries combined with a carefully tuned robot controller lead to learning quadruped locomotion in only 20 minutes in the real world. We evaluate our approach on several indoor and outdoor terrains which are known to be challenging for classical model-"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2208.07860","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2022-08-16T17:37:36Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"821fa3b6561255bd9d47c426e392b29e7eddab078804575f9958daa0223944d8","abstract_canon_sha256":"23b5eb5b00f2cbb48e27f253eaf536b8a1a18abd397c3ebd5d224333e9a260a4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:49:09.957390Z","signature_b64":"MCB3r8FDxgt6JmnRCM5sLzOtvP4iJa29JIFf03OGOt8BjoxiDe1bE5aE8J814TpdEInX/PmjyxAOFZEuXx33CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bf12a4ef24d4ea647ffff87e874d86f05856c7d9509c1a3506fe71d9cbb0750a","last_reissued_at":"2026-07-05T04:49:09.956952Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:49:09.956952Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Walk in the Park: Learning to Walk in 20 Minutes With Model-Free Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.RO","authors_text":"Ilya Kostrikov, Laura Smith, Sergey Levine","submitted_at":"2022-08-16T17:37:36Z","abstract_excerpt":"Deep reinforcement learning is a promising approach to learning policies in uncontrolled environments that do not require domain knowledge. Unfortunately, due to sample inefficiency, deep RL applications have primarily focused on simulated environments. In this work, we demonstrate that the recent advancements in machine learning algorithms and libraries combined with a carefully tuned robot controller lead to learning quadruped locomotion in only 20 minutes in the real world. We evaluate our approach on several indoor and outdoor terrains which are known to be challenging for classical model-"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2208.07860","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2208.07860/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2208.07860","created_at":"2026-07-05T04:49:09.957015+00:00"},{"alias_kind":"arxiv_version","alias_value":"2208.07860v1","created_at":"2026-07-05T04:49:09.957015+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2208.07860","created_at":"2026-07-05T04:49:09.957015+00:00"},{"alias_kind":"pith_short_12","alias_value":"X4JKJ3ZE2TVG","created_at":"2026-07-05T04:49:09.957015+00:00"},{"alias_kind":"pith_short_16","alias_value":"X4JKJ3ZE2TVGI777","created_at":"2026-07-05T04:49:09.957015+00:00"},{"alias_kind":"pith_short_8","alias_value":"X4JKJ3ZE","created_at":"2026-07-05T04:49:09.957015+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23640","citing_title":"Learning Process Rewards via Success Visitation Matching for Efficient RL","ref_index":75,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09595","citing_title":"Neuromorphic Reinforcement Learning for Quadruped Locomotion Control on Uneven Terrain","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10236","citing_title":"When Does Non-Uniform Replay Matter in Reinforcement Learning?","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2601.14617","citing_title":"UniCon: A Unified System for Efficient Robot Learning Transfers","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2602.09580","citing_title":"SERNF: Sample-Efficient Real-World Dexterous Policy Fine-Tuning via Action-Chunked Critics and Normalizing Flows","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2603.15759","citing_title":"Simulation Distillation: Pretraining World Models in Simulation for Rapid Real-World Adaptation","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14350","citing_title":"Distributionally Robust Multi-Task Reinforcement Learning via Adaptive Task Sampling","ref_index":108,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10236","citing_title":"When Does Non-Uniform Replay Matter in Reinforcement Learning?","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09595","citing_title":"Neuromorphic Reinforcement Learning for Quadruped Locomotion Control on Uneven Terrain","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10236","citing_title":"When Does Non-Uniform Replay Matter in Reinforcement Learning?","ref_index":36,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/X4JKJ3ZE2TVGI7777B7IOTMG6B","json":"https://pith.science/pith/X4JKJ3ZE2TVGI7777B7IOTMG6B.json","graph_json":"https://pith.science/api/pith-number/X4JKJ3ZE2TVGI7777B7IOTMG6B/graph.json","events_json":"https://pith.science/api/pith-number/X4JKJ3ZE2TVGI7777B7IOTMG6B/events.json","paper":"https://pith.science/paper/X4JKJ3ZE"},"agent_actions":{"view_html":"https://pith.science/pith/X4JKJ3ZE2TVGI7777B7IOTMG6B","download_json":"https://pith.science/pith/X4JKJ3ZE2TVGI7777B7IOTMG6B.json","view_paper":"https://pith.science/paper/X4JKJ3ZE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2208.07860&json=true","fetch_graph":"https://pith.science/api/pith-number/X4JKJ3ZE2TVGI7777B7IOTMG6B/graph.json","fetch_events":"https://pith.science/api/pith-number/X4JKJ3ZE2TVGI7777B7IOTMG6B/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/X4JKJ3ZE2TVGI7777B7IOTMG6B/action/timestamp_anchor","attest_storage":"https://pith.science/pith/X4JKJ3ZE2TVGI7777B7IOTMG6B/action/storage_attestation","attest_author":"https://pith.science/pith/X4JKJ3ZE2TVGI7777B7IOTMG6B/action/author_attestation","sign_citation":"https://pith.science/pith/X4JKJ3ZE2TVGI7777B7IOTMG6B/action/citation_signature","submit_replication":"https://pith.science/pith/X4JKJ3ZE2TVGI7777B7IOTMG6B/action/replication_record"}},"created_at":"2026-07-05T04:49:09.957015+00:00","updated_at":"2026-07-05T04:49:09.957015+00:00"}