{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:C67GIGIU4BOORCCHOT6XIENKVI","short_pith_number":"pith:C67GIGIU","schema_version":"1.0","canonical_sha256":"17be641914e05ce8884774fd7411aaaa1df67ae78f8059f8f0f683c54f3df9dd","source":{"kind":"arxiv","id":"2405.01718","version":1},"attestation_state":"computed","paper":{"title":"Robust Risk-Sensitive Reinforcement Learning with Conditional Value-at-Risk","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Lifeng Lai, Xinyi Ni","submitted_at":"2024-05-02T20:28:49Z","abstract_excerpt":"Robust Markov Decision Processes (RMDPs) have received significant research interest, offering an alternative to standard Markov Decision Processes (MDPs) that often assume fixed transition probabilities. RMDPs address this by optimizing for the worst-case scenarios within ambiguity sets. While earlier studies on RMDPs have largely centered on risk-neutral reinforcement learning (RL), with the goal of minimizing expected total discounted costs, in this paper, we analyze the robustness of CVaR-based risk-sensitive RL under RMDP. Firstly, we consider predetermined ambiguity sets. Based on the co"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.01718","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-02T20:28:49Z","cross_cats_sorted":["math.OC","stat.ML"],"title_canon_sha256":"41c7d07a76fc29e79668ecfa7db3a30830f75a9b14f29277ba9fbea0e6ca7e61","abstract_canon_sha256":"2d81e04223ca70046b2f0f798c45a027cd1a0fac7749b2c25f3191f75dafe4cd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:15:01.166836Z","signature_b64":"Y7tNgQ59XBTD8y/t+1nTEhci/EJgXYwa9y4sfgyDp2sqWIYyW0GhEaE9wCSAe3z1r/DnvfHv/CWgvU160h9SAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"17be641914e05ce8884774fd7411aaaa1df67ae78f8059f8f0f683c54f3df9dd","last_reissued_at":"2026-07-05T08:15:01.166431Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:15:01.166431Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Robust Risk-Sensitive Reinforcement Learning with Conditional Value-at-Risk","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Lifeng Lai, Xinyi Ni","submitted_at":"2024-05-02T20:28:49Z","abstract_excerpt":"Robust Markov Decision Processes (RMDPs) have received significant research interest, offering an alternative to standard Markov Decision Processes (MDPs) that often assume fixed transition probabilities. RMDPs address this by optimizing for the worst-case scenarios within ambiguity sets. While earlier studies on RMDPs have largely centered on risk-neutral reinforcement learning (RL), with the goal of minimizing expected total discounted costs, in this paper, we analyze the robustness of CVaR-based risk-sensitive RL under RMDP. Firstly, we consider predetermined ambiguity sets. Based on the co"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.01718","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.01718/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.01718","created_at":"2026-07-05T08:15:01.166485+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.01718v1","created_at":"2026-07-05T08:15:01.166485+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.01718","created_at":"2026-07-05T08:15:01.166485+00:00"},{"alias_kind":"pith_short_12","alias_value":"C67GIGIU4BOO","created_at":"2026-07-05T08:15:01.166485+00:00"},{"alias_kind":"pith_short_16","alias_value":"C67GIGIU4BOORCCH","created_at":"2026-07-05T08:15:01.166485+00:00"},{"alias_kind":"pith_short_8","alias_value":"C67GIGIU","created_at":"2026-07-05T08:15:01.166485+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/C67GIGIU4BOORCCHOT6XIENKVI","json":"https://pith.science/pith/C67GIGIU4BOORCCHOT6XIENKVI.json","graph_json":"https://pith.science/api/pith-number/C67GIGIU4BOORCCHOT6XIENKVI/graph.json","events_json":"https://pith.science/api/pith-number/C67GIGIU4BOORCCHOT6XIENKVI/events.json","paper":"https://pith.science/paper/C67GIGIU"},"agent_actions":{"view_html":"https://pith.science/pith/C67GIGIU4BOORCCHOT6XIENKVI","download_json":"https://pith.science/pith/C67GIGIU4BOORCCHOT6XIENKVI.json","view_paper":"https://pith.science/paper/C67GIGIU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.01718&json=true","fetch_graph":"https://pith.science/api/pith-number/C67GIGIU4BOORCCHOT6XIENKVI/graph.json","fetch_events":"https://pith.science/api/pith-number/C67GIGIU4BOORCCHOT6XIENKVI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/C67GIGIU4BOORCCHOT6XIENKVI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/C67GIGIU4BOORCCHOT6XIENKVI/action/storage_attestation","attest_author":"https://pith.science/pith/C67GIGIU4BOORCCHOT6XIENKVI/action/author_attestation","sign_citation":"https://pith.science/pith/C67GIGIU4BOORCCHOT6XIENKVI/action/citation_signature","submit_replication":"https://pith.science/pith/C67GIGIU4BOORCCHOT6XIENKVI/action/replication_record"}},"created_at":"2026-07-05T08:15:01.166485+00:00","updated_at":"2026-07-05T08:15:01.166485+00:00"}