{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:FKEO474PDH4JHF47HPHO5U53KN","short_pith_number":"pith:FKEO474P","schema_version":"1.0","canonical_sha256":"2a88ee7f8f19f893979f3bceeed3bb534352843f249884c636067a3f2d956da8","source":{"kind":"arxiv","id":"2310.17634","version":1},"attestation_state":"computed","paper":{"title":"Grow Your Limits: Continuous Improvement with Real-World RL for Robotic Locomotion","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.RO","authors_text":"Laura Smith, Sergey Levine, Yunhao Cao","submitted_at":"2023-10-26T17:51:46Z","abstract_excerpt":"Deep reinforcement learning (RL) can enable robots to autonomously acquire complex behaviors, such as legged locomotion. However, RL in the real world is complicated by constraints on efficiency, safety, and overall training stability, which limits its practical applicability. We present APRL, a policy regularization framework that modulates the robot's exploration over the course of training, striking a balance between flexible improvement potential and focused, efficient exploration. APRL enables a quadrupedal robot to efficiently learn to walk entirely in the real world within minutes and c"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.17634","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2023-10-26T17:51:46Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"3a02f6e72f165bc9e2367839e4a0f4f1fee666c24c92ecbe8377227f0c5e281f","abstract_canon_sha256":"a0f753a197f166825958301a15e7ae84fbd9c6af39030cd61cf9476f503eafe1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:05:37.196646Z","signature_b64":"+XK4TRqbw0OXk8N7F1k9tIbepTd8BM1t/AFU4e4JDjfx5r50+7oOkTVaUGLG/+IXtlm1w1NhlqRDoFSjyy8TDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2a88ee7f8f19f893979f3bceeed3bb534352843f249884c636067a3f2d956da8","last_reissued_at":"2026-07-05T07:05:37.196161Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:05:37.196161Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Grow Your Limits: Continuous Improvement with Real-World RL for Robotic Locomotion","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.RO","authors_text":"Laura Smith, Sergey Levine, Yunhao Cao","submitted_at":"2023-10-26T17:51:46Z","abstract_excerpt":"Deep reinforcement learning (RL) can enable robots to autonomously acquire complex behaviors, such as legged locomotion. However, RL in the real world is complicated by constraints on efficiency, safety, and overall training stability, which limits its practical applicability. We present APRL, a policy regularization framework that modulates the robot's exploration over the course of training, striking a balance between flexible improvement potential and focused, efficient exploration. APRL enables a quadrupedal robot to efficiently learn to walk entirely in the real world within minutes and c"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.17634","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.17634/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.17634","created_at":"2026-07-05T07:05:37.196218+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.17634v1","created_at":"2026-07-05T07:05:37.196218+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.17634","created_at":"2026-07-05T07:05:37.196218+00:00"},{"alias_kind":"pith_short_12","alias_value":"FKEO474PDH4J","created_at":"2026-07-05T07:05:37.196218+00:00"},{"alias_kind":"pith_short_16","alias_value":"FKEO474PDH4JHF47","created_at":"2026-07-05T07:05:37.196218+00:00"},{"alias_kind":"pith_short_8","alias_value":"FKEO474P","created_at":"2026-07-05T07:05:37.196218+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.10429","citing_title":"Real Time Control of Tandem-Wing Experimental Platform Using Concerto Reinforcement Learning","ref_index":16,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FKEO474PDH4JHF47HPHO5U53KN","json":"https://pith.science/pith/FKEO474PDH4JHF47HPHO5U53KN.json","graph_json":"https://pith.science/api/pith-number/FKEO474PDH4JHF47HPHO5U53KN/graph.json","events_json":"https://pith.science/api/pith-number/FKEO474PDH4JHF47HPHO5U53KN/events.json","paper":"https://pith.science/paper/FKEO474P"},"agent_actions":{"view_html":"https://pith.science/pith/FKEO474PDH4JHF47HPHO5U53KN","download_json":"https://pith.science/pith/FKEO474PDH4JHF47HPHO5U53KN.json","view_paper":"https://pith.science/paper/FKEO474P","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.17634&json=true","fetch_graph":"https://pith.science/api/pith-number/FKEO474PDH4JHF47HPHO5U53KN/graph.json","fetch_events":"https://pith.science/api/pith-number/FKEO474PDH4JHF47HPHO5U53KN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FKEO474PDH4JHF47HPHO5U53KN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FKEO474PDH4JHF47HPHO5U53KN/action/storage_attestation","attest_author":"https://pith.science/pith/FKEO474PDH4JHF47HPHO5U53KN/action/author_attestation","sign_citation":"https://pith.science/pith/FKEO474PDH4JHF47HPHO5U53KN/action/citation_signature","submit_replication":"https://pith.science/pith/FKEO474PDH4JHF47HPHO5U53KN/action/replication_record"}},"created_at":"2026-07-05T07:05:37.196218+00:00","updated_at":"2026-07-05T07:05:37.196218+00:00"}