{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:MSJIUTZBTVDVZI63XRLRDNKJSH","short_pith_number":"pith:MSJIUTZB","schema_version":"1.0","canonical_sha256":"64928a4f219d475ca3dbbc5711b54991c38f77573c81e5ec33a03bcba6408c2e","source":{"kind":"arxiv","id":"2012.06644","version":2},"attestation_state":"computed","paper":{"title":"Regularizing Action Policies for Smooth Control with Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.SY","eess.SY"],"primary_cat":"cs.RO","authors_text":"Bassel Mabsout, Kate Saenko, Renato Mancuso, Siddharth Mysore","submitted_at":"2020-12-11T21:35:24Z","abstract_excerpt":"A critical problem with the practical utility of controllers trained with deep Reinforcement Learning (RL) is the notable lack of smoothness in the actions learned by the RL policies. This trend often presents itself in the form of control signal oscillation and can result in poor control, high power consumption, and undue system wear. We introduce Conditioning for Action Policy Smoothness (CAPS), an effective yet intuitive regularization on action policies, which offers consistent improvement in the smoothness of the learned state-to-action mappings of neural network controllers, reflected in"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2012.06644","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2020-12-11T21:35:24Z","cross_cats_sorted":["cs.LG","cs.SY","eess.SY"],"title_canon_sha256":"7310a45a293923e6843e2b18d4a658f80c6a8d7dfd52573e3b60d47d05922bbe","abstract_canon_sha256":"295d6774577cac129081bec861f2fde6175eead08077b9ec62c681d4e885ae7d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:43:43.165206Z","signature_b64":"9Z3kC2hS2HhsxWkvybvObSQYheIzPNM+kBjwUbAWugQRq1WH8d+RiTzdzTqZxHrIuauMOJVmLdumqTHE7jPQCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"64928a4f219d475ca3dbbc5711b54991c38f77573c81e5ec33a03bcba6408c2e","last_reissued_at":"2026-07-05T02:43:43.164680Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:43:43.164680Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Regularizing Action Policies for Smooth Control with Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.SY","eess.SY"],"primary_cat":"cs.RO","authors_text":"Bassel Mabsout, Kate Saenko, Renato Mancuso, Siddharth Mysore","submitted_at":"2020-12-11T21:35:24Z","abstract_excerpt":"A critical problem with the practical utility of controllers trained with deep Reinforcement Learning (RL) is the notable lack of smoothness in the actions learned by the RL policies. This trend often presents itself in the form of control signal oscillation and can result in poor control, high power consumption, and undue system wear. We introduce Conditioning for Action Policy Smoothness (CAPS), an effective yet intuitive regularization on action policies, which offers consistent improvement in the smoothness of the learned state-to-action mappings of neural network controllers, reflected in"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2012.06644","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2012.06644/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2012.06644","created_at":"2026-07-05T02:43:43.164744+00:00"},{"alias_kind":"arxiv_version","alias_value":"2012.06644v2","created_at":"2026-07-05T02:43:43.164744+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2012.06644","created_at":"2026-07-05T02:43:43.164744+00:00"},{"alias_kind":"pith_short_12","alias_value":"MSJIUTZBTVDV","created_at":"2026-07-05T02:43:43.164744+00:00"},{"alias_kind":"pith_short_16","alias_value":"MSJIUTZBTVDVZI63","created_at":"2026-07-05T02:43:43.164744+00:00"},{"alias_kind":"pith_short_8","alias_value":"MSJIUTZB","created_at":"2026-07-05T02:43:43.164744+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.19592","citing_title":"Implicit Action Chunking for Smooth Continuous Control","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17744","citing_title":"Input-Side Variance Suppression under Non-Normal Transient Amplification in Continuous-Control Reinforcement Learning","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MSJIUTZBTVDVZI63XRLRDNKJSH","json":"https://pith.science/pith/MSJIUTZBTVDVZI63XRLRDNKJSH.json","graph_json":"https://pith.science/api/pith-number/MSJIUTZBTVDVZI63XRLRDNKJSH/graph.json","events_json":"https://pith.science/api/pith-number/MSJIUTZBTVDVZI63XRLRDNKJSH/events.json","paper":"https://pith.science/paper/MSJIUTZB"},"agent_actions":{"view_html":"https://pith.science/pith/MSJIUTZBTVDVZI63XRLRDNKJSH","download_json":"https://pith.science/pith/MSJIUTZBTVDVZI63XRLRDNKJSH.json","view_paper":"https://pith.science/paper/MSJIUTZB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2012.06644&json=true","fetch_graph":"https://pith.science/api/pith-number/MSJIUTZBTVDVZI63XRLRDNKJSH/graph.json","fetch_events":"https://pith.science/api/pith-number/MSJIUTZBTVDVZI63XRLRDNKJSH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MSJIUTZBTVDVZI63XRLRDNKJSH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MSJIUTZBTVDVZI63XRLRDNKJSH/action/storage_attestation","attest_author":"https://pith.science/pith/MSJIUTZBTVDVZI63XRLRDNKJSH/action/author_attestation","sign_citation":"https://pith.science/pith/MSJIUTZBTVDVZI63XRLRDNKJSH/action/citation_signature","submit_replication":"https://pith.science/pith/MSJIUTZBTVDVZI63XRLRDNKJSH/action/replication_record"}},"created_at":"2026-07-05T02:43:43.164744+00:00","updated_at":"2026-07-05T02:43:43.164744+00:00"}