{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:XMOPVDUHSYSSOVP7DG7FH5UIVD","short_pith_number":"pith:XMOPVDUH","schema_version":"1.0","canonical_sha256":"bb1cfa8e8796252755ff19be53f688a8d8f47c6e657abc26d41555f2dd9c9f3e","source":{"kind":"arxiv","id":"2201.11965","version":4},"attestation_state":"computed","paper":{"title":"Provably Efficient Primal-Dual Reinforcement Learning for CMDPs with Non-stationary Objectives and Constraints","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Javad Lavaei, Yuhao Ding","submitted_at":"2022-01-28T07:18:29Z","abstract_excerpt":"We consider primal-dual-based reinforcement learning (RL) in episodic constrained Markov decision processes (CMDPs) with non-stationary objectives and constraints, which plays a central role in ensuring the safety of RL in time-varying environments. In this problem, the reward/utility functions and the state transition functions are both allowed to vary arbitrarily over time as long as their cumulative variations do not exceed certain known variation budgets. Designing safe RL algorithms in time-varying environments is particularly challenging because of the need to integrate the constraint vi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2201.11965","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-01-28T07:18:29Z","cross_cats_sorted":[],"title_canon_sha256":"7d265f369c751dac8a15c482ea3ec1b701c6842187b2eb646920b36c0c9b75e9","abstract_canon_sha256":"614fd7d25123d27c664eff6b690f8064dc52df835101b93f380b83f4fb9fc13d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:17:25.057571Z","signature_b64":"N/8dVjJ4I91Lx2rsnXfJdLsGXHCyrZsIWEq4dptRNrLbAzG+csMvssjGpphJbufxp143oGdtTvn5YL46kbTyDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bb1cfa8e8796252755ff19be53f688a8d8f47c6e657abc26d41555f2dd9c9f3e","last_reissued_at":"2026-07-05T05:17:25.057126Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:17:25.057126Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Provably Efficient Primal-Dual Reinforcement Learning for CMDPs with Non-stationary Objectives and Constraints","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Javad Lavaei, Yuhao Ding","submitted_at":"2022-01-28T07:18:29Z","abstract_excerpt":"We consider primal-dual-based reinforcement learning (RL) in episodic constrained Markov decision processes (CMDPs) with non-stationary objectives and constraints, which plays a central role in ensuring the safety of RL in time-varying environments. In this problem, the reward/utility functions and the state transition functions are both allowed to vary arbitrarily over time as long as their cumulative variations do not exceed certain known variation budgets. Designing safe RL algorithms in time-varying environments is particularly challenging because of the need to integrate the constraint vi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2201.11965","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2201.11965/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2201.11965","created_at":"2026-07-05T05:17:25.057204+00:00"},{"alias_kind":"arxiv_version","alias_value":"2201.11965v4","created_at":"2026-07-05T05:17:25.057204+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2201.11965","created_at":"2026-07-05T05:17:25.057204+00:00"},{"alias_kind":"pith_short_12","alias_value":"XMOPVDUHSYSS","created_at":"2026-07-05T05:17:25.057204+00:00"},{"alias_kind":"pith_short_16","alias_value":"XMOPVDUHSYSSOVP7","created_at":"2026-07-05T05:17:25.057204+00:00"},{"alias_kind":"pith_short_8","alias_value":"XMOPVDUH","created_at":"2026-07-05T05:17:25.057204+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.21841","citing_title":"An Optimistic Algorithm for online CMDPS with Anytime Adversarial Constraints","ref_index":13,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XMOPVDUHSYSSOVP7DG7FH5UIVD","json":"https://pith.science/pith/XMOPVDUHSYSSOVP7DG7FH5UIVD.json","graph_json":"https://pith.science/api/pith-number/XMOPVDUHSYSSOVP7DG7FH5UIVD/graph.json","events_json":"https://pith.science/api/pith-number/XMOPVDUHSYSSOVP7DG7FH5UIVD/events.json","paper":"https://pith.science/paper/XMOPVDUH"},"agent_actions":{"view_html":"https://pith.science/pith/XMOPVDUHSYSSOVP7DG7FH5UIVD","download_json":"https://pith.science/pith/XMOPVDUHSYSSOVP7DG7FH5UIVD.json","view_paper":"https://pith.science/paper/XMOPVDUH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2201.11965&json=true","fetch_graph":"https://pith.science/api/pith-number/XMOPVDUHSYSSOVP7DG7FH5UIVD/graph.json","fetch_events":"https://pith.science/api/pith-number/XMOPVDUHSYSSOVP7DG7FH5UIVD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XMOPVDUHSYSSOVP7DG7FH5UIVD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XMOPVDUHSYSSOVP7DG7FH5UIVD/action/storage_attestation","attest_author":"https://pith.science/pith/XMOPVDUHSYSSOVP7DG7FH5UIVD/action/author_attestation","sign_citation":"https://pith.science/pith/XMOPVDUHSYSSOVP7DG7FH5UIVD/action/citation_signature","submit_replication":"https://pith.science/pith/XMOPVDUHSYSSOVP7DG7FH5UIVD/action/replication_record"}},"created_at":"2026-07-05T05:17:25.057204+00:00","updated_at":"2026-07-05T05:17:25.057204+00:00"}