{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:FQSDQZFNTNJPUBHNV7MOBX2TXI","short_pith_number":"pith:FQSDQZFN","canonical_record":{"source":{"id":"2508.01883","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-08-03T18:35:55Z","cross_cats_sorted":[],"title_canon_sha256":"fe8d0f635a8c4dde87446ed5ba1f1cf8d86e26cf7743842cbde37988bf25452e","abstract_canon_sha256":"5f57f976a72dd40d564cc00596104878fba6af44ed70ddc75e8771cf38940f0d"},"schema_version":"1.0"},"canonical_sha256":"2c243864ad9b52fa04edafd8e0df53ba06fb2232dead43cb0f02026001f5349d","source":{"kind":"arxiv","id":"2508.01883","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2508.01883","created_at":"2026-07-05T11:49:34Z"},{"alias_kind":"arxiv_version","alias_value":"2508.01883v2","created_at":"2026-07-05T11:49:34Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.01883","created_at":"2026-07-05T11:49:34Z"},{"alias_kind":"pith_short_12","alias_value":"FQSDQZFNTNJP","created_at":"2026-07-05T11:49:34Z"},{"alias_kind":"pith_short_16","alias_value":"FQSDQZFNTNJPUBHN","created_at":"2026-07-05T11:49:34Z"},{"alias_kind":"pith_short_8","alias_value":"FQSDQZFN","created_at":"2026-07-05T11:49:34Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:FQSDQZFNTNJPUBHNV7MOBX2TXI","target":"record","payload":{"canonical_record":{"source":{"id":"2508.01883","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-08-03T18:35:55Z","cross_cats_sorted":[],"title_canon_sha256":"fe8d0f635a8c4dde87446ed5ba1f1cf8d86e26cf7743842cbde37988bf25452e","abstract_canon_sha256":"5f57f976a72dd40d564cc00596104878fba6af44ed70ddc75e8771cf38940f0d"},"schema_version":"1.0"},"canonical_sha256":"2c243864ad9b52fa04edafd8e0df53ba06fb2232dead43cb0f02026001f5349d","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:49:34.398818Z","signature_b64":"s1r1OFZ1GQ6bIVsNwwi3vZufhOohuP6iX6pVg+gPdSNNMXzeK7rFqYvx217RIrywcso19ZZBFiH5PexuL439Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2c243864ad9b52fa04edafd8e0df53ba06fb2232dead43cb0f02026001f5349d","last_reissued_at":"2026-07-05T11:49:34.398368Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:49:34.398368Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2508.01883","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:49:34Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"YQ9BWR0xTrxjabUeN0nQSblbpWmLvTHUyehcFI4oqKbtTvxw/PFFEsSIr+0TbaCaP6d25B1oK+GQyFgFTtEKCA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T23:21:37.025724Z"},"content_sha256":"8f76c48aa577b8ef7218eb203bd0f6d1d2e7a1de3bad65fbd10b18a7998e0cee","schema_version":"1.0","event_id":"sha256:8f76c48aa577b8ef7218eb203bd0f6d1d2e7a1de3bad65fbd10b18a7998e0cee"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:FQSDQZFNTNJPUBHNV7MOBX2TXI","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Proactive Constrained Policy Optimization with Preemptive Penalty","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Guoqing Liu, Haifeng Zhang, Jun Wang, Ning Yang, Pengyu Wang, Pin Lv","submitted_at":"2025-08-03T18:35:55Z","abstract_excerpt":"Safe Reinforcement Learning (RL) often faces significant issues such as constraint violations and instability, necessitating the use of constrained policy optimization, which seeks optimal policies while ensuring adherence to specific constraints like safety. Typically, constrained optimization problems are addressed by the Lagrangian method, a post-violation remedial approach that may result in oscillations and overshoots. Motivated by this, we propose a novel method named Proactive Constrained Policy Optimization (PCPO) that incorporates a preemptive penalty mechanism. This mechanism integra"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.01883","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.01883/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:49:34Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"lDsQiNfXYF7VsP4DOaZM18T1o0pSdexU2kFbV6GdjLi/YRIAn1hdgywtHFx2s8/TYwVISiaHdUUndRQRd4UeDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T23:21:37.026056Z"},"content_sha256":"716882eea2663ae96e2d2fa45af9449bd41f9da59b396c00d8ddeee6992d6f53","schema_version":"1.0","event_id":"sha256:716882eea2663ae96e2d2fa45af9449bd41f9da59b396c00d8ddeee6992d6f53"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/FQSDQZFNTNJPUBHNV7MOBX2TXI/bundle.json","state_url":"https://pith.science/pith/FQSDQZFNTNJPUBHNV7MOBX2TXI/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/FQSDQZFNTNJPUBHNV7MOBX2TXI/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-04T23:21:37Z","links":{"resolver":"https://pith.science/pith/FQSDQZFNTNJPUBHNV7MOBX2TXI","bundle":"https://pith.science/pith/FQSDQZFNTNJPUBHNV7MOBX2TXI/bundle.json","state":"https://pith.science/pith/FQSDQZFNTNJPUBHNV7MOBX2TXI/state.json","well_known_bundle":"https://pith.science/.well-known/pith/FQSDQZFNTNJPUBHNV7MOBX2TXI/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:FQSDQZFNTNJPUBHNV7MOBX2TXI","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"5f57f976a72dd40d564cc00596104878fba6af44ed70ddc75e8771cf38940f0d","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-08-03T18:35:55Z","title_canon_sha256":"fe8d0f635a8c4dde87446ed5ba1f1cf8d86e26cf7743842cbde37988bf25452e"},"schema_version":"1.0","source":{"id":"2508.01883","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2508.01883","created_at":"2026-07-05T11:49:34Z"},{"alias_kind":"arxiv_version","alias_value":"2508.01883v2","created_at":"2026-07-05T11:49:34Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.01883","created_at":"2026-07-05T11:49:34Z"},{"alias_kind":"pith_short_12","alias_value":"FQSDQZFNTNJP","created_at":"2026-07-05T11:49:34Z"},{"alias_kind":"pith_short_16","alias_value":"FQSDQZFNTNJPUBHN","created_at":"2026-07-05T11:49:34Z"},{"alias_kind":"pith_short_8","alias_value":"FQSDQZFN","created_at":"2026-07-05T11:49:34Z"}],"graph_snapshots":[{"event_id":"sha256:716882eea2663ae96e2d2fa45af9449bd41f9da59b396c00d8ddeee6992d6f53","target":"graph","created_at":"2026-07-05T11:49:34Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2508.01883/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Safe Reinforcement Learning (RL) often faces significant issues such as constraint violations and instability, necessitating the use of constrained policy optimization, which seeks optimal policies while ensuring adherence to specific constraints like safety. Typically, constrained optimization problems are addressed by the Lagrangian method, a post-violation remedial approach that may result in oscillations and overshoots. Motivated by this, we propose a novel method named Proactive Constrained Policy Optimization (PCPO) that incorporates a preemptive penalty mechanism. This mechanism integra","authors_text":"Guoqing Liu, Haifeng Zhang, Jun Wang, Ning Yang, Pengyu Wang, Pin Lv","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-08-03T18:35:55Z","title":"Proactive Constrained Policy Optimization with Preemptive Penalty"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.01883","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:8f76c48aa577b8ef7218eb203bd0f6d1d2e7a1de3bad65fbd10b18a7998e0cee","target":"record","created_at":"2026-07-05T11:49:34Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"5f57f976a72dd40d564cc00596104878fba6af44ed70ddc75e8771cf38940f0d","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-08-03T18:35:55Z","title_canon_sha256":"fe8d0f635a8c4dde87446ed5ba1f1cf8d86e26cf7743842cbde37988bf25452e"},"schema_version":"1.0","source":{"id":"2508.01883","kind":"arxiv","version":2}},"canonical_sha256":"2c243864ad9b52fa04edafd8e0df53ba06fb2232dead43cb0f02026001f5349d","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"2c243864ad9b52fa04edafd8e0df53ba06fb2232dead43cb0f02026001f5349d","first_computed_at":"2026-07-05T11:49:34.398368Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:49:34.398368Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"s1r1OFZ1GQ6bIVsNwwi3vZufhOohuP6iX6pVg+gPdSNNMXzeK7rFqYvx217RIrywcso19ZZBFiH5PexuL439Cg==","signature_status":"signed_v1","signed_at":"2026-07-05T11:49:34.398818Z","signed_message":"canonical_sha256_bytes"},"source_id":"2508.01883","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:8f76c48aa577b8ef7218eb203bd0f6d1d2e7a1de3bad65fbd10b18a7998e0cee","sha256:716882eea2663ae96e2d2fa45af9449bd41f9da59b396c00d8ddeee6992d6f53"],"state_sha256":"dcd4bc9688999ff4c0bfe1c5feeb35b2c1e588e3dd325244ce1e30cf11e2e4db"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"1KWovsxGF7PIuLW2VIntEgVb5M5DsMlViEI8UmWj0VEav2lr4ENSHP3iaDfh4pZfITAKfeqv6EkPyiYPtPq5Cg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-04T23:21:37.029080Z","bundle_sha256":"440ed4aa26d2924fc565f323f3782f0c5534be34bd08af222208392a0e6d1224"}}