{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:JE3LL2IIHDZSSQHWKAHVV6VDUK","short_pith_number":"pith:JE3LL2II","schema_version":"1.0","canonical_sha256":"4936b5e90838f32940f6500f5afaa3a2a9599efe93da069a6cccfe950c19a341","source":{"kind":"arxiv","id":"2102.11941","version":2},"attestation_state":"computed","paper":{"title":"State Augmented Constrained Reinforcement Learning: Overcoming the Limitations of Learning with Rewards","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.RO","math.OC"],"primary_cat":"cs.LG","authors_text":"Alejandro Ribeiro, Luiz F. O. Chamon, Miguel Calvo-Fullana, Santiago Paternain","submitted_at":"2021-02-23T21:07:35Z","abstract_excerpt":"A common formulation of constrained reinforcement learning involves multiple rewards that must individually accumulate to given thresholds. In this class of problems, we show a simple example in which the desired optimal policy cannot be induced by any weighted linear combination of rewards. Hence, there exist constrained reinforcement learning problems for which neither regularized nor classical primal-dual methods yield optimal policies. This work addresses this shortcoming by augmenting the state with Lagrange multipliers and reinterpreting primal-dual methods as the portion of the dynamics"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2102.11941","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-02-23T21:07:35Z","cross_cats_sorted":["cs.RO","math.OC"],"title_canon_sha256":"1b55a1873405aef81a77becb6f4b58afb0d142c2247661bc1cb050376535ea4e","abstract_canon_sha256":"190eae3b728dc09a6973a4a153b72e4f6dfdf58562d12a4c899241dc7fa622a7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:52:42.174344Z","signature_b64":"x6OFrHPXD7frtYay9XD2t6JmaG/F/rj9kgjMpo/mKOl3gp+qzL3O3gxlKh4Bu8A2yEVg/tS9dmhUqFCgmAQsDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4936b5e90838f32940f6500f5afaa3a2a9599efe93da069a6cccfe950c19a341","last_reissued_at":"2026-07-05T06:52:42.173931Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:52:42.173931Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"State Augmented Constrained Reinforcement Learning: Overcoming the Limitations of Learning with Rewards","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.RO","math.OC"],"primary_cat":"cs.LG","authors_text":"Alejandro Ribeiro, Luiz F. O. Chamon, Miguel Calvo-Fullana, Santiago Paternain","submitted_at":"2021-02-23T21:07:35Z","abstract_excerpt":"A common formulation of constrained reinforcement learning involves multiple rewards that must individually accumulate to given thresholds. In this class of problems, we show a simple example in which the desired optimal policy cannot be induced by any weighted linear combination of rewards. Hence, there exist constrained reinforcement learning problems for which neither regularized nor classical primal-dual methods yield optimal policies. This work addresses this shortcoming by augmenting the state with Lagrange multipliers and reinterpreting primal-dual methods as the portion of the dynamics"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2102.11941","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2102.11941/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2102.11941","created_at":"2026-07-05T06:52:42.173983+00:00"},{"alias_kind":"arxiv_version","alias_value":"2102.11941v2","created_at":"2026-07-05T06:52:42.173983+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2102.11941","created_at":"2026-07-05T06:52:42.173983+00:00"},{"alias_kind":"pith_short_12","alias_value":"JE3LL2IIHDZS","created_at":"2026-07-05T06:52:42.173983+00:00"},{"alias_kind":"pith_short_16","alias_value":"JE3LL2IIHDZSSQHW","created_at":"2026-07-05T06:52:42.173983+00:00"},{"alias_kind":"pith_short_8","alias_value":"JE3LL2II","created_at":"2026-07-05T06:52:42.173983+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.14002","citing_title":"Operator Splitting for Convex Constrained Markov Decision Processes","ref_index":14,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JE3LL2IIHDZSSQHWKAHVV6VDUK","json":"https://pith.science/pith/JE3LL2IIHDZSSQHWKAHVV6VDUK.json","graph_json":"https://pith.science/api/pith-number/JE3LL2IIHDZSSQHWKAHVV6VDUK/graph.json","events_json":"https://pith.science/api/pith-number/JE3LL2IIHDZSSQHWKAHVV6VDUK/events.json","paper":"https://pith.science/paper/JE3LL2II"},"agent_actions":{"view_html":"https://pith.science/pith/JE3LL2IIHDZSSQHWKAHVV6VDUK","download_json":"https://pith.science/pith/JE3LL2IIHDZSSQHWKAHVV6VDUK.json","view_paper":"https://pith.science/paper/JE3LL2II","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2102.11941&json=true","fetch_graph":"https://pith.science/api/pith-number/JE3LL2IIHDZSSQHWKAHVV6VDUK/graph.json","fetch_events":"https://pith.science/api/pith-number/JE3LL2IIHDZSSQHWKAHVV6VDUK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JE3LL2IIHDZSSQHWKAHVV6VDUK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JE3LL2IIHDZSSQHWKAHVV6VDUK/action/storage_attestation","attest_author":"https://pith.science/pith/JE3LL2IIHDZSSQHWKAHVV6VDUK/action/author_attestation","sign_citation":"https://pith.science/pith/JE3LL2IIHDZSSQHWKAHVV6VDUK/action/citation_signature","submit_replication":"https://pith.science/pith/JE3LL2IIHDZSSQHWKAHVV6VDUK/action/replication_record"}},"created_at":"2026-07-05T06:52:42.173983+00:00","updated_at":"2026-07-05T06:52:42.173983+00:00"}