{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:TU3KYV7OXQVYAX43DRZBT6N5XT","short_pith_number":"pith:TU3KYV7O","schema_version":"1.0","canonical_sha256":"9d36ac57eebc2b805f9b1c7219f9bdbccdd34cfb4da4fe93f94500235883736c","source":{"kind":"arxiv","id":"2403.14508","version":1},"attestation_state":"computed","paper":{"title":"Constrained Reinforcement Learning with Smoothed Log Barrier Function","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.SY","eess.SY"],"primary_cat":"cs.LG","authors_text":"Baohe Zhang, Joschka B\\\"odecker, Lilli Frison, Thomas Brox, Yuan Zhang","submitted_at":"2024-03-21T16:02:52Z","abstract_excerpt":"Reinforcement Learning (RL) has been widely applied to many control tasks and substantially improved the performances compared to conventional control methods in many domains where the reward function is well defined. However, for many real-world problems, it is often more convenient to formulate optimization problems in terms of rewards and constraints simultaneously. Optimizing such constrained problems via reward shaping can be difficult as it requires tedious manual tuning of reward functions with several interacting terms. Recent formulations which include constraints mostly require a pre"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.14508","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2024-03-21T16:02:52Z","cross_cats_sorted":["cs.AI","cs.SY","eess.SY"],"title_canon_sha256":"127279dc8432f3a4abf7689400018462f0d693c8d385d4aea76b32b7b32fd813","abstract_canon_sha256":"6ec93019d59dae1195de849afc3f4052f872022ec4742710364b0d7755b6ba68"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:59:10.133057Z","signature_b64":"Jmv+LRGI+FsCj7pJIduWu9lFp0VqC3qCcZ0DmhBiGBwvo0lgqOVkRKAteSwielFYE5lLGdHMjYN7pykxd0MgBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9d36ac57eebc2b805f9b1c7219f9bdbccdd34cfb4da4fe93f94500235883736c","last_reissued_at":"2026-07-05T07:59:10.132561Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:59:10.132561Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Constrained Reinforcement Learning with Smoothed Log Barrier Function","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.SY","eess.SY"],"primary_cat":"cs.LG","authors_text":"Baohe Zhang, Joschka B\\\"odecker, Lilli Frison, Thomas Brox, Yuan Zhang","submitted_at":"2024-03-21T16:02:52Z","abstract_excerpt":"Reinforcement Learning (RL) has been widely applied to many control tasks and substantially improved the performances compared to conventional control methods in many domains where the reward function is well defined. However, for many real-world problems, it is often more convenient to formulate optimization problems in terms of rewards and constraints simultaneously. Optimizing such constrained problems via reward shaping can be difficult as it requires tedious manual tuning of reward functions with several interacting terms. Recent formulations which include constraints mostly require a pre"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.14508","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.14508/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.14508","created_at":"2026-07-05T07:59:10.132623+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.14508v1","created_at":"2026-07-05T07:59:10.132623+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.14508","created_at":"2026-07-05T07:59:10.132623+00:00"},{"alias_kind":"pith_short_12","alias_value":"TU3KYV7OXQVY","created_at":"2026-07-05T07:59:10.132623+00:00"},{"alias_kind":"pith_short_16","alias_value":"TU3KYV7OXQVYAX43","created_at":"2026-07-05T07:59:10.132623+00:00"},{"alias_kind":"pith_short_8","alias_value":"TU3KYV7O","created_at":"2026-07-05T07:59:10.132623+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2508.09128","citing_title":"A Review On Safe Reinforcement Learning Using Lyapunov and Barrier Functions","ref_index":86,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27955","citing_title":"GUI Agents with Reinforcement Learning: Toward Digital Inhabitants","ref_index":82,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23576","citing_title":"CAPSULE: Control-Theoretic Action Perturbations for Safe Uncertainty-Aware Reinforcement Learning","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00393","citing_title":"Model-Based Reinforcement Learning with Double Oracle Efficiency in Policy Optimization and Offline Estimation","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TU3KYV7OXQVYAX43DRZBT6N5XT","json":"https://pith.science/pith/TU3KYV7OXQVYAX43DRZBT6N5XT.json","graph_json":"https://pith.science/api/pith-number/TU3KYV7OXQVYAX43DRZBT6N5XT/graph.json","events_json":"https://pith.science/api/pith-number/TU3KYV7OXQVYAX43DRZBT6N5XT/events.json","paper":"https://pith.science/paper/TU3KYV7O"},"agent_actions":{"view_html":"https://pith.science/pith/TU3KYV7OXQVYAX43DRZBT6N5XT","download_json":"https://pith.science/pith/TU3KYV7OXQVYAX43DRZBT6N5XT.json","view_paper":"https://pith.science/paper/TU3KYV7O","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.14508&json=true","fetch_graph":"https://pith.science/api/pith-number/TU3KYV7OXQVYAX43DRZBT6N5XT/graph.json","fetch_events":"https://pith.science/api/pith-number/TU3KYV7OXQVYAX43DRZBT6N5XT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TU3KYV7OXQVYAX43DRZBT6N5XT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TU3KYV7OXQVYAX43DRZBT6N5XT/action/storage_attestation","attest_author":"https://pith.science/pith/TU3KYV7OXQVYAX43DRZBT6N5XT/action/author_attestation","sign_citation":"https://pith.science/pith/TU3KYV7OXQVYAX43DRZBT6N5XT/action/citation_signature","submit_replication":"https://pith.science/pith/TU3KYV7OXQVYAX43DRZBT6N5XT/action/replication_record"}},"created_at":"2026-07-05T07:59:10.132623+00:00","updated_at":"2026-07-05T07:59:10.132623+00:00"}