{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:X5HZOBMU2FTGIX4CGMOBNBMMHY","short_pith_number":"pith:X5HZOBMU","schema_version":"1.0","canonical_sha256":"bf4f970594d166645f82331c16858c3e1bd45f3f9899ee377688077f1004dae2","source":{"kind":"arxiv","id":"2202.07565","version":1},"attestation_state":"computed","paper":{"title":"CUP: A Conservative Update Policy Algorithm for Safe Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Gang Pan, Jiaming Ji, Juntao Dai, Long Yang, Pengfei Li, Yu Zhang","submitted_at":"2022-02-15T16:49:28Z","abstract_excerpt":"Safe reinforcement learning (RL) is still very challenging since it requires the agent to consider both return maximization and safe exploration. In this paper, we propose CUP, a Conservative Update Policy algorithm with a theoretical safety guarantee. We derive the CUP based on the new proposed performance bounds and surrogate functions. Although using bounds as surrogate functions to design safe RL algorithms have appeared in some existing works, we develop them at least three aspects: (i) We provide a rigorous theoretical analysis to extend the surrogate functions to generalized advantage e"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2202.07565","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-02-15T16:49:28Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"c0e62b809c4885835d85629f54a5df4368888cc7c4980b8b262a32e05c87ea98","abstract_canon_sha256":"baeaa71367ee84484a9a3c5d61388af98fbf98579d5f073715227a0372bdfd43"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:57:01.385121Z","signature_b64":"VlwTQYLUBuPukojUCAJ4JNIe/4UvadLJZm2HJmH/iQZY+PtOLmrCSGZMmEAhuOT2eX5RMJXjfk7vzF8P8eV3BA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bf4f970594d166645f82331c16858c3e1bd45f3f9899ee377688077f1004dae2","last_reissued_at":"2026-07-05T03:57:01.384669Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:57:01.384669Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CUP: A Conservative Update Policy Algorithm for Safe Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Gang Pan, Jiaming Ji, Juntao Dai, Long Yang, Pengfei Li, Yu Zhang","submitted_at":"2022-02-15T16:49:28Z","abstract_excerpt":"Safe reinforcement learning (RL) is still very challenging since it requires the agent to consider both return maximization and safe exploration. In this paper, we propose CUP, a Conservative Update Policy algorithm with a theoretical safety guarantee. We derive the CUP based on the new proposed performance bounds and surrogate functions. Although using bounds as surrogate functions to design safe RL algorithms have appeared in some existing works, we develop them at least three aspects: (i) We provide a rigorous theoretical analysis to extend the surrogate functions to generalized advantage e"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2202.07565","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2202.07565/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2202.07565","created_at":"2026-07-05T03:57:01.384727+00:00"},{"alias_kind":"arxiv_version","alias_value":"2202.07565v1","created_at":"2026-07-05T03:57:01.384727+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2202.07565","created_at":"2026-07-05T03:57:01.384727+00:00"},{"alias_kind":"pith_short_12","alias_value":"X5HZOBMU2FTG","created_at":"2026-07-05T03:57:01.384727+00:00"},{"alias_kind":"pith_short_16","alias_value":"X5HZOBMU2FTGIX4C","created_at":"2026-07-05T03:57:01.384727+00:00"},{"alias_kind":"pith_short_8","alias_value":"X5HZOBMU","created_at":"2026-07-05T03:57:01.384727+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.14246","citing_title":"Action-Conditioned Risk Gating for Safety-Critical Control under Partial Observability","ref_index":33,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/X5HZOBMU2FTGIX4CGMOBNBMMHY","json":"https://pith.science/pith/X5HZOBMU2FTGIX4CGMOBNBMMHY.json","graph_json":"https://pith.science/api/pith-number/X5HZOBMU2FTGIX4CGMOBNBMMHY/graph.json","events_json":"https://pith.science/api/pith-number/X5HZOBMU2FTGIX4CGMOBNBMMHY/events.json","paper":"https://pith.science/paper/X5HZOBMU"},"agent_actions":{"view_html":"https://pith.science/pith/X5HZOBMU2FTGIX4CGMOBNBMMHY","download_json":"https://pith.science/pith/X5HZOBMU2FTGIX4CGMOBNBMMHY.json","view_paper":"https://pith.science/paper/X5HZOBMU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2202.07565&json=true","fetch_graph":"https://pith.science/api/pith-number/X5HZOBMU2FTGIX4CGMOBNBMMHY/graph.json","fetch_events":"https://pith.science/api/pith-number/X5HZOBMU2FTGIX4CGMOBNBMMHY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/X5HZOBMU2FTGIX4CGMOBNBMMHY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/X5HZOBMU2FTGIX4CGMOBNBMMHY/action/storage_attestation","attest_author":"https://pith.science/pith/X5HZOBMU2FTGIX4CGMOBNBMMHY/action/author_attestation","sign_citation":"https://pith.science/pith/X5HZOBMU2FTGIX4CGMOBNBMMHY/action/citation_signature","submit_replication":"https://pith.science/pith/X5HZOBMU2FTGIX4CGMOBNBMMHY/action/replication_record"}},"created_at":"2026-07-05T03:57:01.384727+00:00","updated_at":"2026-07-05T03:57:01.384727+00:00"}