{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:2WLGZDERKDCXBX37UK3PBXMZ5D","short_pith_number":"pith:2WLGZDER","schema_version":"1.0","canonical_sha256":"d5966c8c9150c570df7fa2b6f0dd99e8fc692107ecb9b17786cf80daf85c39ba","source":{"kind":"arxiv","id":"2208.06721","version":2},"attestation_state":"computed","paper":{"title":"Lyapunov Design for Robust and Efficient Robotic Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SY","eess.SY"],"primary_cat":"cs.RO","authors_text":"Ayush Agrawal, Fernando Castaneda, Koushil Sreenath, Shankar Sastry, Tyler Westenbroek","submitted_at":"2022-08-13T20:04:17Z","abstract_excerpt":"Recent advances in the reinforcement learning (RL) literature have enabled roboticists to automatically train complex policies in simulated environments. However, due to the poor sample complexity of these methods, solving RL problems using real-world data remains a challenging problem. This paper introduces a novel cost-shaping method which aims to reduce the number of samples needed to learn a stabilizing controller. The method adds a term involving a Control Lyapunov Function (CLF) -- an `energy-like' function from the model-based control literature -- to typical cost formulations. Theoreti"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2208.06721","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2022-08-13T20:04:17Z","cross_cats_sorted":["cs.SY","eess.SY"],"title_canon_sha256":"10c51f00540dc6222969466252c6ecf48520ab1f7a19ef787a054892cc8b2bdd","abstract_canon_sha256":"3c6df4f78c24f2e8849007f2b83084ef4c1a46459021fba667221dd8b6b71a87"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:17:12.014257Z","signature_b64":"BVgtkyZchJ3GkmGfsS2sVm1wPYRRLGgwJ70asUND9aO/3gOgkj0GokJO4Dy2rjhc/l6i7Sh9FiwaAdnt7GtTAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d5966c8c9150c570df7fa2b6f0dd99e8fc692107ecb9b17786cf80daf85c39ba","last_reissued_at":"2026-07-05T05:17:12.013853Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:17:12.013853Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Lyapunov Design for Robust and Efficient Robotic Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SY","eess.SY"],"primary_cat":"cs.RO","authors_text":"Ayush Agrawal, Fernando Castaneda, Koushil Sreenath, Shankar Sastry, Tyler Westenbroek","submitted_at":"2022-08-13T20:04:17Z","abstract_excerpt":"Recent advances in the reinforcement learning (RL) literature have enabled roboticists to automatically train complex policies in simulated environments. However, due to the poor sample complexity of these methods, solving RL problems using real-world data remains a challenging problem. This paper introduces a novel cost-shaping method which aims to reduce the number of samples needed to learn a stabilizing controller. The method adds a term involving a Control Lyapunov Function (CLF) -- an `energy-like' function from the model-based control literature -- to typical cost formulations. Theoreti"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2208.06721","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2208.06721/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2208.06721","created_at":"2026-07-05T05:17:12.013919+00:00"},{"alias_kind":"arxiv_version","alias_value":"2208.06721v2","created_at":"2026-07-05T05:17:12.013919+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2208.06721","created_at":"2026-07-05T05:17:12.013919+00:00"},{"alias_kind":"pith_short_12","alias_value":"2WLGZDERKDCX","created_at":"2026-07-05T05:17:12.013919+00:00"},{"alias_kind":"pith_short_16","alias_value":"2WLGZDERKDCXBX37","created_at":"2026-07-05T05:17:12.013919+00:00"},{"alias_kind":"pith_short_8","alias_value":"2WLGZDER","created_at":"2026-07-05T05:17:12.013919+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.15971","citing_title":"OHP-RL: Online Human Preference as Guidance in Reinforcement Learning for Robot Manipulation","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22145","citing_title":"Zero-shot Transfer of Reinforcement Learning Control Policies for the Swing-Up and Stabilization of a Cart-Pole System","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26392","citing_title":"MPC-Injection: Biasing Off-Policy Locomotion RL Toward Controller-Induced Behavior Basins","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15971","citing_title":"OHP-RL: Online Human Preference as Guidance in Reinforcement Learning for Robot Manipulation","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2508.09128","citing_title":"A Review On Safe Reinforcement Learning Using Lyapunov and Barrier Functions","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2603.15759","citing_title":"Simulation Distillation: Pretraining World Models in Simulation for Rapid Real-World Adaptation","ref_index":52,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2WLGZDERKDCXBX37UK3PBXMZ5D","json":"https://pith.science/pith/2WLGZDERKDCXBX37UK3PBXMZ5D.json","graph_json":"https://pith.science/api/pith-number/2WLGZDERKDCXBX37UK3PBXMZ5D/graph.json","events_json":"https://pith.science/api/pith-number/2WLGZDERKDCXBX37UK3PBXMZ5D/events.json","paper":"https://pith.science/paper/2WLGZDER"},"agent_actions":{"view_html":"https://pith.science/pith/2WLGZDERKDCXBX37UK3PBXMZ5D","download_json":"https://pith.science/pith/2WLGZDERKDCXBX37UK3PBXMZ5D.json","view_paper":"https://pith.science/paper/2WLGZDER","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2208.06721&json=true","fetch_graph":"https://pith.science/api/pith-number/2WLGZDERKDCXBX37UK3PBXMZ5D/graph.json","fetch_events":"https://pith.science/api/pith-number/2WLGZDERKDCXBX37UK3PBXMZ5D/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2WLGZDERKDCXBX37UK3PBXMZ5D/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2WLGZDERKDCXBX37UK3PBXMZ5D/action/storage_attestation","attest_author":"https://pith.science/pith/2WLGZDERKDCXBX37UK3PBXMZ5D/action/author_attestation","sign_citation":"https://pith.science/pith/2WLGZDERKDCXBX37UK3PBXMZ5D/action/citation_signature","submit_replication":"https://pith.science/pith/2WLGZDERKDCXBX37UK3PBXMZ5D/action/replication_record"}},"created_at":"2026-07-05T05:17:12.013919+00:00","updated_at":"2026-07-05T05:17:12.013919+00:00"}