{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:4ML7U3LTPUMTMHFJ7YGT3FXHOJ","short_pith_number":"pith:4ML7U3LT","schema_version":"1.0","canonical_sha256":"e317fa6d737d19361ca9fe0d3d96e77241702cd5e5c64a6cb42861ebfb0577e6","source":{"kind":"arxiv","id":"2506.00700","version":2},"attestation_state":"computed","paper":{"title":"Central Path Proximal Policy Optimization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Johannes M\\\"uller, Nico Scherf, Nikola Milosevic","submitted_at":"2025-05-31T20:14:29Z","abstract_excerpt":"In constrained Markov decision processes, enforcing constraints during training is often thought of as decreasing the final return. Recently, it was shown that constraints can be incorporated directly into the policy geometry, yielding an optimization trajectory close to the central path of a barrier method, which does not compromise final return. Building on this idea, we introduce Central Path Proximal Policy Optimization (C3PO), a simple modification of the PPO loss that produces policy iterates, that stay close to the central path of the constrained optimization problem. Compared to existi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.00700","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-31T20:14:29Z","cross_cats_sorted":[],"title_canon_sha256":"74a9e8a7c4df03ef4c90092b734935c5935aed90aac3779be703fec71d1c6d01","abstract_canon_sha256":"cd5b7727502edeaa3d2aa4310b5207cb93a4023906faf2e6c208e7b7df0639f4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:54:31.049820Z","signature_b64":"Hth13F/uMUxC29ORF0R+5IbFmErcOXn1Cpb6/QO4Dtfy7ZdZIBwl0fE60xHk0bV2I/TYjZsXtTFcGC4+LNncBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e317fa6d737d19361ca9fe0d3d96e77241702cd5e5c64a6cb42861ebfb0577e6","last_reissued_at":"2026-07-05T11:54:31.049375Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:54:31.049375Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Central Path Proximal Policy Optimization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Johannes M\\\"uller, Nico Scherf, Nikola Milosevic","submitted_at":"2025-05-31T20:14:29Z","abstract_excerpt":"In constrained Markov decision processes, enforcing constraints during training is often thought of as decreasing the final return. Recently, it was shown that constraints can be incorporated directly into the policy geometry, yielding an optimization trajectory close to the central path of a barrier method, which does not compromise final return. Building on this idea, we introduce Central Path Proximal Policy Optimization (C3PO), a simple modification of the PPO loss that produces policy iterates, that stay close to the central path of the constrained optimization problem. Compared to existi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.00700","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.00700/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.00700","created_at":"2026-07-05T11:54:31.049431+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.00700v2","created_at":"2026-07-05T11:54:31.049431+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.00700","created_at":"2026-07-05T11:54:31.049431+00:00"},{"alias_kind":"pith_short_12","alias_value":"4ML7U3LTPUMT","created_at":"2026-07-05T11:54:31.049431+00:00"},{"alias_kind":"pith_short_16","alias_value":"4ML7U3LTPUMTMHFJ","created_at":"2026-07-05T11:54:31.049431+00:00"},{"alias_kind":"pith_short_8","alias_value":"4ML7U3LT","created_at":"2026-07-05T11:54:31.049431+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.18578","citing_title":"Bounded Ratio Reinforcement Learning","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4ML7U3LTPUMTMHFJ7YGT3FXHOJ","json":"https://pith.science/pith/4ML7U3LTPUMTMHFJ7YGT3FXHOJ.json","graph_json":"https://pith.science/api/pith-number/4ML7U3LTPUMTMHFJ7YGT3FXHOJ/graph.json","events_json":"https://pith.science/api/pith-number/4ML7U3LTPUMTMHFJ7YGT3FXHOJ/events.json","paper":"https://pith.science/paper/4ML7U3LT"},"agent_actions":{"view_html":"https://pith.science/pith/4ML7U3LTPUMTMHFJ7YGT3FXHOJ","download_json":"https://pith.science/pith/4ML7U3LTPUMTMHFJ7YGT3FXHOJ.json","view_paper":"https://pith.science/paper/4ML7U3LT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.00700&json=true","fetch_graph":"https://pith.science/api/pith-number/4ML7U3LTPUMTMHFJ7YGT3FXHOJ/graph.json","fetch_events":"https://pith.science/api/pith-number/4ML7U3LTPUMTMHFJ7YGT3FXHOJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4ML7U3LTPUMTMHFJ7YGT3FXHOJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4ML7U3LTPUMTMHFJ7YGT3FXHOJ/action/storage_attestation","attest_author":"https://pith.science/pith/4ML7U3LTPUMTMHFJ7YGT3FXHOJ/action/author_attestation","sign_citation":"https://pith.science/pith/4ML7U3LTPUMTMHFJ7YGT3FXHOJ/action/citation_signature","submit_replication":"https://pith.science/pith/4ML7U3LTPUMTMHFJ7YGT3FXHOJ/action/replication_record"}},"created_at":"2026-07-05T11:54:31.049431+00:00","updated_at":"2026-07-05T11:54:31.049431+00:00"}