{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:DY4S6F4PPHG623OIOUGLA7N36P","short_pith_number":"pith:DY4S6F4P","schema_version":"1.0","canonical_sha256":"1e392f178f79cded6dc8750cb07dbbf3c0cd1cb524873ec5a4f4628c679acadc","source":{"kind":"arxiv","id":"2306.11700","version":2},"attestation_state":"computed","paper":{"title":"Last-Iterate Convergent Policy Gradient Primal-Dual Methods for Constrained MDPs","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.LG","cs.SY","eess.SY"],"primary_cat":"math.OC","authors_text":"Alejandro Ribeiro, Chen-Yu Wei, Dongsheng Ding, Kaiqing Zhang","submitted_at":"2023-06-20T17:27:31Z","abstract_excerpt":"We study the problem of computing an optimal policy of an infinite-horizon discounted constrained Markov decision process (constrained MDP). Despite the popularity of Lagrangian-based policy search methods used in practice, the oscillation of policy iterates in these methods has not been fully understood, bringing out issues such as violation of constraints and sensitivity to hyper-parameters. To fill this gap, we employ the Lagrangian method to cast a constrained MDP into a constrained saddle-point problem in which max/min players correspond to primal/dual variables, respectively, and develop"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.11700","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"math.OC","submitted_at":"2023-06-20T17:27:31Z","cross_cats_sorted":["cs.LG","cs.SY","eess.SY"],"title_canon_sha256":"2c47f317309ac95c26ebeb7965ed366431c84e1c383ee58ff52de2ce2cb5e168","abstract_canon_sha256":"89bd978a0cd89bf62b5feea479d06400302181dc4527c8e5a0f3da256f9649a2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:34:15.556878Z","signature_b64":"WX3s4SKc5tTv4OhJMe+8VlACOYnY1Y5KeG+50kHqBfwtkBOwwDjVfdS5gJ6vYu5YIFTlx0jo4fbLaWQ1e7XaCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1e392f178f79cded6dc8750cb07dbbf3c0cd1cb524873ec5a4f4628c679acadc","last_reissued_at":"2026-07-05T07:34:15.556421Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:34:15.556421Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Last-Iterate Convergent Policy Gradient Primal-Dual Methods for Constrained MDPs","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.LG","cs.SY","eess.SY"],"primary_cat":"math.OC","authors_text":"Alejandro Ribeiro, Chen-Yu Wei, Dongsheng Ding, Kaiqing Zhang","submitted_at":"2023-06-20T17:27:31Z","abstract_excerpt":"We study the problem of computing an optimal policy of an infinite-horizon discounted constrained Markov decision process (constrained MDP). Despite the popularity of Lagrangian-based policy search methods used in practice, the oscillation of policy iterates in these methods has not been fully understood, bringing out issues such as violation of constraints and sensitivity to hyper-parameters. To fill this gap, we employ the Lagrangian method to cast a constrained MDP into a constrained saddle-point problem in which max/min players correspond to primal/dual variables, respectively, and develop"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.11700","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.11700/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.11700","created_at":"2026-07-05T07:34:15.556484+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.11700v2","created_at":"2026-07-05T07:34:15.556484+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.11700","created_at":"2026-07-05T07:34:15.556484+00:00"},{"alias_kind":"pith_short_12","alias_value":"DY4S6F4PPHG6","created_at":"2026-07-05T07:34:15.556484+00:00"},{"alias_kind":"pith_short_16","alias_value":"DY4S6F4PPHG623OI","created_at":"2026-07-05T07:34:15.556484+00:00"},{"alias_kind":"pith_short_8","alias_value":"DY4S6F4P","created_at":"2026-07-05T07:34:15.556484+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.14243","citing_title":"Optimistic Policy Learning under Pessimistic Adversaries with Regret and Violation Guarantees","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DY4S6F4PPHG623OIOUGLA7N36P","json":"https://pith.science/pith/DY4S6F4PPHG623OIOUGLA7N36P.json","graph_json":"https://pith.science/api/pith-number/DY4S6F4PPHG623OIOUGLA7N36P/graph.json","events_json":"https://pith.science/api/pith-number/DY4S6F4PPHG623OIOUGLA7N36P/events.json","paper":"https://pith.science/paper/DY4S6F4P"},"agent_actions":{"view_html":"https://pith.science/pith/DY4S6F4PPHG623OIOUGLA7N36P","download_json":"https://pith.science/pith/DY4S6F4PPHG623OIOUGLA7N36P.json","view_paper":"https://pith.science/paper/DY4S6F4P","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.11700&json=true","fetch_graph":"https://pith.science/api/pith-number/DY4S6F4PPHG623OIOUGLA7N36P/graph.json","fetch_events":"https://pith.science/api/pith-number/DY4S6F4PPHG623OIOUGLA7N36P/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DY4S6F4PPHG623OIOUGLA7N36P/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DY4S6F4PPHG623OIOUGLA7N36P/action/storage_attestation","attest_author":"https://pith.science/pith/DY4S6F4PPHG623OIOUGLA7N36P/action/author_attestation","sign_citation":"https://pith.science/pith/DY4S6F4PPHG623OIOUGLA7N36P/action/citation_signature","submit_replication":"https://pith.science/pith/DY4S6F4PPHG623OIOUGLA7N36P/action/replication_record"}},"created_at":"2026-07-05T07:34:15.556484+00:00","updated_at":"2026-07-05T07:34:15.556484+00:00"}