{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:AALOQM6Z2KYU63PLKTWJ3I7BWJ","short_pith_number":"pith:AALOQM6Z","schema_version":"1.0","canonical_sha256":"0016e833d9d2b14f6deb54ec9da3e1b27a1cf18ce6c0b2ef45f6e111ef8f9261","source":{"kind":"arxiv","id":"2010.04870","version":1},"attestation_state":"computed","paper":{"title":"Robust Constrained-MDPs: Soft-Constrained Robust Policy Optimization under Model Uncertainty","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Jeroen van Baar, Mouhacine Benosman, Reazul Hasan Russel","submitted_at":"2020-10-10T01:53:37Z","abstract_excerpt":"In this paper, we focus on the problem of robustifying reinforcement learning (RL) algorithms with respect to model uncertainties. Indeed, in the framework of model-based RL, we propose to merge the theory of constrained Markov decision process (CMDP), with the theory of robust Markov decision process (RMDP), leading to a formulation of robust constrained-MDPs (RCMDP). This formulation, simple in essence, allows us to design RL algorithms that are robust in performance, and provides constraint satisfaction guarantees, with respect to uncertainties in the system's states transition probabilitie"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2010.04870","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-10-10T01:53:37Z","cross_cats_sorted":[],"title_canon_sha256":"8eb359fd52d11f6ec751e78bbed6e2c7ad8466f145a27b596c51d1c11c4c71ad","abstract_canon_sha256":"78748146145eeb976e97e115e443d4bbf3b280aa7647a0fef671f0587e11e3dc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:42:00.173765Z","signature_b64":"8nBjRELkkVSZDTGiAEb/Q7ka1KrZbvMtBY4xO5DuyW2NCr2XVcoxTGNK1eneuDOnoq3HZFiRjzTcVbrno6XwBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0016e833d9d2b14f6deb54ec9da3e1b27a1cf18ce6c0b2ef45f6e111ef8f9261","last_reissued_at":"2026-07-05T01:42:00.173365Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:42:00.173365Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Robust Constrained-MDPs: Soft-Constrained Robust Policy Optimization under Model Uncertainty","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Jeroen van Baar, Mouhacine Benosman, Reazul Hasan Russel","submitted_at":"2020-10-10T01:53:37Z","abstract_excerpt":"In this paper, we focus on the problem of robustifying reinforcement learning (RL) algorithms with respect to model uncertainties. Indeed, in the framework of model-based RL, we propose to merge the theory of constrained Markov decision process (CMDP), with the theory of robust Markov decision process (RMDP), leading to a formulation of robust constrained-MDPs (RCMDP). This formulation, simple in essence, allows us to design RL algorithms that are robust in performance, and provides constraint satisfaction guarantees, with respect to uncertainties in the system's states transition probabilitie"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2010.04870","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2010.04870/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2010.04870","created_at":"2026-07-05T01:42:00.173427+00:00"},{"alias_kind":"arxiv_version","alias_value":"2010.04870v1","created_at":"2026-07-05T01:42:00.173427+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2010.04870","created_at":"2026-07-05T01:42:00.173427+00:00"},{"alias_kind":"pith_short_12","alias_value":"AALOQM6Z2KYU","created_at":"2026-07-05T01:42:00.173427+00:00"},{"alias_kind":"pith_short_16","alias_value":"AALOQM6Z2KYU63PL","created_at":"2026-07-05T01:42:00.173427+00:00"},{"alias_kind":"pith_short_8","alias_value":"AALOQM6Z","created_at":"2026-07-05T01:42:00.173427+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.05350","citing_title":"Characterization and Analysis of Emergency Landing Flight Envelopes with Graded Safety Specifications","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00270","citing_title":"Robust Shielding for Safe Reinforcement Learning","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2408.16286","citing_title":"Near-Optimal Policy Identification in Robust Constrained Markov Decision Processes via Epigraph Form","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21177","citing_title":"Revisiting Subgradient Dominance in Robust MDPs: Counterexamples, Hardness, and Sufficient Conditions","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14243","citing_title":"Optimistic Policy Learning under Pessimistic Adversaries with Regret and Violation Guarantees","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AALOQM6Z2KYU63PLKTWJ3I7BWJ","json":"https://pith.science/pith/AALOQM6Z2KYU63PLKTWJ3I7BWJ.json","graph_json":"https://pith.science/api/pith-number/AALOQM6Z2KYU63PLKTWJ3I7BWJ/graph.json","events_json":"https://pith.science/api/pith-number/AALOQM6Z2KYU63PLKTWJ3I7BWJ/events.json","paper":"https://pith.science/paper/AALOQM6Z"},"agent_actions":{"view_html":"https://pith.science/pith/AALOQM6Z2KYU63PLKTWJ3I7BWJ","download_json":"https://pith.science/pith/AALOQM6Z2KYU63PLKTWJ3I7BWJ.json","view_paper":"https://pith.science/paper/AALOQM6Z","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2010.04870&json=true","fetch_graph":"https://pith.science/api/pith-number/AALOQM6Z2KYU63PLKTWJ3I7BWJ/graph.json","fetch_events":"https://pith.science/api/pith-number/AALOQM6Z2KYU63PLKTWJ3I7BWJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AALOQM6Z2KYU63PLKTWJ3I7BWJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AALOQM6Z2KYU63PLKTWJ3I7BWJ/action/storage_attestation","attest_author":"https://pith.science/pith/AALOQM6Z2KYU63PLKTWJ3I7BWJ/action/author_attestation","sign_citation":"https://pith.science/pith/AALOQM6Z2KYU63PLKTWJ3I7BWJ/action/citation_signature","submit_replication":"https://pith.science/pith/AALOQM6Z2KYU63PLKTWJ3I7BWJ/action/replication_record"}},"created_at":"2026-07-05T01:42:00.173427+00:00","updated_at":"2026-07-05T01:42:00.173427+00:00"}