{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:ZR7CVTWRU6UPWC6KUHW64YGQQ2","short_pith_number":"pith:ZR7CVTWR","schema_version":"1.0","canonical_sha256":"cc7e2aced1a7a8fb0bcaa1edee60d086b465f0fafb67e5a5ad45c941d5b1c1ef","source":{"kind":"arxiv","id":"2003.05555","version":6},"attestation_state":"computed","paper":{"title":"Provably Efficient Model-Free Algorithm for MDPs with Peak Constraints","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.SY","eess.SY","stat.ML"],"primary_cat":"math.OC","authors_text":"Ather Gattami, Qinbo Bai, Vaneet Aggarwal","submitted_at":"2020-03-11T23:23:29Z","abstract_excerpt":"In the optimization of dynamic systems, the variables typically have constraints. Such problems can be modeled as a Constrained Markov Decision Process (CMDP). This paper considers the peak Constrained Markov Decision Process (PCMDP), where the agent chooses the policy to maximize total reward in the finite horizon as well as satisfy constraints at each epoch with probability 1. We propose a model-free algorithm that converts PCMDP problem to an unconstrained problem and a Q-learning based approach is applied. We define the concept of probably approximately correct (PAC) to the proposed PCMDP "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2003.05555","kind":"arxiv","version":6},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"math.OC","submitted_at":"2020-03-11T23:23:29Z","cross_cats_sorted":["cs.LG","cs.SY","eess.SY","stat.ML"],"title_canon_sha256":"d77af39c96709f2c5cff1c38a905525843b748fab37c60cd69b1ea72f2808388","abstract_canon_sha256":"4c4dc4f956607eb7a824f258172131f0ea65b515944345f0187a8b073e0b96e5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:31:19.413448Z","signature_b64":"a6ADVikMP7XEX2IO78sANBVWFEK1gJtUuFcRhd3gPLIL2O9273lQrjSc2mtZbabKIyNxYaFRxbMZOd7UJdgwCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cc7e2aced1a7a8fb0bcaa1edee60d086b465f0fafb67e5a5ad45c941d5b1c1ef","last_reissued_at":"2026-07-05T04:31:19.412940Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:31:19.412940Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Provably Efficient Model-Free Algorithm for MDPs with Peak Constraints","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.SY","eess.SY","stat.ML"],"primary_cat":"math.OC","authors_text":"Ather Gattami, Qinbo Bai, Vaneet Aggarwal","submitted_at":"2020-03-11T23:23:29Z","abstract_excerpt":"In the optimization of dynamic systems, the variables typically have constraints. Such problems can be modeled as a Constrained Markov Decision Process (CMDP). This paper considers the peak Constrained Markov Decision Process (PCMDP), where the agent chooses the policy to maximize total reward in the finite horizon as well as satisfy constraints at each epoch with probability 1. We propose a model-free algorithm that converts PCMDP problem to an unconstrained problem and a Q-learning based approach is applied. We define the concept of probably approximately correct (PAC) to the proposed PCMDP "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2003.05555","kind":"arxiv","version":6},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2003.05555/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2003.05555","created_at":"2026-07-05T04:31:19.412999+00:00"},{"alias_kind":"arxiv_version","alias_value":"2003.05555v6","created_at":"2026-07-05T04:31:19.412999+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2003.05555","created_at":"2026-07-05T04:31:19.412999+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZR7CVTWRU6UP","created_at":"2026-07-05T04:31:19.412999+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZR7CVTWRU6UPWC6K","created_at":"2026-07-05T04:31:19.412999+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZR7CVTWR","created_at":"2026-07-05T04:31:19.412999+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.08290","citing_title":"Toward Optimal Regret in Robust Pricing: Decoupling Corruption and Time","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10519","citing_title":"Online Resource Allocation With General Constraints","ref_index":20,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZR7CVTWRU6UPWC6KUHW64YGQQ2","json":"https://pith.science/pith/ZR7CVTWRU6UPWC6KUHW64YGQQ2.json","graph_json":"https://pith.science/api/pith-number/ZR7CVTWRU6UPWC6KUHW64YGQQ2/graph.json","events_json":"https://pith.science/api/pith-number/ZR7CVTWRU6UPWC6KUHW64YGQQ2/events.json","paper":"https://pith.science/paper/ZR7CVTWR"},"agent_actions":{"view_html":"https://pith.science/pith/ZR7CVTWRU6UPWC6KUHW64YGQQ2","download_json":"https://pith.science/pith/ZR7CVTWRU6UPWC6KUHW64YGQQ2.json","view_paper":"https://pith.science/paper/ZR7CVTWR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2003.05555&json=true","fetch_graph":"https://pith.science/api/pith-number/ZR7CVTWRU6UPWC6KUHW64YGQQ2/graph.json","fetch_events":"https://pith.science/api/pith-number/ZR7CVTWRU6UPWC6KUHW64YGQQ2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZR7CVTWRU6UPWC6KUHW64YGQQ2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZR7CVTWRU6UPWC6KUHW64YGQQ2/action/storage_attestation","attest_author":"https://pith.science/pith/ZR7CVTWRU6UPWC6KUHW64YGQQ2/action/author_attestation","sign_citation":"https://pith.science/pith/ZR7CVTWRU6UPWC6KUHW64YGQQ2/action/citation_signature","submit_replication":"https://pith.science/pith/ZR7CVTWRU6UPWC6KUHW64YGQQ2/action/replication_record"}},"created_at":"2026-07-05T04:31:19.412999+00:00","updated_at":"2026-07-05T04:31:19.412999+00:00"}