{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:FQSDQZFNTNJPUBHNV7MOBX2TXI","short_pith_number":"pith:FQSDQZFN","schema_version":"1.0","canonical_sha256":"2c243864ad9b52fa04edafd8e0df53ba06fb2232dead43cb0f02026001f5349d","source":{"kind":"arxiv","id":"2508.01883","version":2},"attestation_state":"computed","paper":{"title":"Proactive Constrained Policy Optimization with Preemptive Penalty","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Guoqing Liu, Haifeng Zhang, Jun Wang, Ning Yang, Pengyu Wang, Pin Lv","submitted_at":"2025-08-03T18:35:55Z","abstract_excerpt":"Safe Reinforcement Learning (RL) often faces significant issues such as constraint violations and instability, necessitating the use of constrained policy optimization, which seeks optimal policies while ensuring adherence to specific constraints like safety. Typically, constrained optimization problems are addressed by the Lagrangian method, a post-violation remedial approach that may result in oscillations and overshoots. Motivated by this, we propose a novel method named Proactive Constrained Policy Optimization (PCPO) that incorporates a preemptive penalty mechanism. This mechanism integra"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.01883","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-08-03T18:35:55Z","cross_cats_sorted":[],"title_canon_sha256":"fe8d0f635a8c4dde87446ed5ba1f1cf8d86e26cf7743842cbde37988bf25452e","abstract_canon_sha256":"5f57f976a72dd40d564cc00596104878fba6af44ed70ddc75e8771cf38940f0d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:49:34.398818Z","signature_b64":"s1r1OFZ1GQ6bIVsNwwi3vZufhOohuP6iX6pVg+gPdSNNMXzeK7rFqYvx217RIrywcso19ZZBFiH5PexuL439Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2c243864ad9b52fa04edafd8e0df53ba06fb2232dead43cb0f02026001f5349d","last_reissued_at":"2026-07-05T11:49:34.398368Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:49:34.398368Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Proactive Constrained Policy Optimization with Preemptive Penalty","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Guoqing Liu, Haifeng Zhang, Jun Wang, Ning Yang, Pengyu Wang, Pin Lv","submitted_at":"2025-08-03T18:35:55Z","abstract_excerpt":"Safe Reinforcement Learning (RL) often faces significant issues such as constraint violations and instability, necessitating the use of constrained policy optimization, which seeks optimal policies while ensuring adherence to specific constraints like safety. Typically, constrained optimization problems are addressed by the Lagrangian method, a post-violation remedial approach that may result in oscillations and overshoots. Motivated by this, we propose a novel method named Proactive Constrained Policy Optimization (PCPO) that incorporates a preemptive penalty mechanism. This mechanism integra"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.01883","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.01883/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.01883","created_at":"2026-07-05T11:49:34.398426+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.01883v2","created_at":"2026-07-05T11:49:34.398426+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.01883","created_at":"2026-07-05T11:49:34.398426+00:00"},{"alias_kind":"pith_short_12","alias_value":"FQSDQZFNTNJP","created_at":"2026-07-05T11:49:34.398426+00:00"},{"alias_kind":"pith_short_16","alias_value":"FQSDQZFNTNJPUBHN","created_at":"2026-07-05T11:49:34.398426+00:00"},{"alias_kind":"pith_short_8","alias_value":"FQSDQZFN","created_at":"2026-07-05T11:49:34.398426+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FQSDQZFNTNJPUBHNV7MOBX2TXI","json":"https://pith.science/pith/FQSDQZFNTNJPUBHNV7MOBX2TXI.json","graph_json":"https://pith.science/api/pith-number/FQSDQZFNTNJPUBHNV7MOBX2TXI/graph.json","events_json":"https://pith.science/api/pith-number/FQSDQZFNTNJPUBHNV7MOBX2TXI/events.json","paper":"https://pith.science/paper/FQSDQZFN"},"agent_actions":{"view_html":"https://pith.science/pith/FQSDQZFNTNJPUBHNV7MOBX2TXI","download_json":"https://pith.science/pith/FQSDQZFNTNJPUBHNV7MOBX2TXI.json","view_paper":"https://pith.science/paper/FQSDQZFN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.01883&json=true","fetch_graph":"https://pith.science/api/pith-number/FQSDQZFNTNJPUBHNV7MOBX2TXI/graph.json","fetch_events":"https://pith.science/api/pith-number/FQSDQZFNTNJPUBHNV7MOBX2TXI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FQSDQZFNTNJPUBHNV7MOBX2TXI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FQSDQZFNTNJPUBHNV7MOBX2TXI/action/storage_attestation","attest_author":"https://pith.science/pith/FQSDQZFNTNJPUBHNV7MOBX2TXI/action/author_attestation","sign_citation":"https://pith.science/pith/FQSDQZFNTNJPUBHNV7MOBX2TXI/action/citation_signature","submit_replication":"https://pith.science/pith/FQSDQZFNTNJPUBHNV7MOBX2TXI/action/replication_record"}},"created_at":"2026-07-05T11:49:34.398426+00:00","updated_at":"2026-07-05T11:49:34.398426+00:00"}