{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2023:I6KL2I5IGOFNTCQFAFBQO56LCZ","short_pith_number":"pith:I6KL2I5I","canonical_record":{"source":{"id":"2312.03618","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"math.OC","submitted_at":"2023-12-06T17:04:08Z","cross_cats_sorted":["cs.GT"],"title_canon_sha256":"8beb380635bb84adfeb60c207dd450f7f704b50a95869857272f0e731c421fce","abstract_canon_sha256":"12912bc1e11c6c68347d645a8f05221cf113a034fc2f61fbdc9d8b3fe1203f25"},"schema_version":"1.0"},"canonical_sha256":"4794bd23a8338ad98a0501430777cb166ae834799e71c1daec491351f50b76a0","source":{"kind":"arxiv","id":"2312.03618","version":3},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2312.03618","created_at":"2026-07-05T10:00:34Z"},{"alias_kind":"arxiv_version","alias_value":"2312.03618v3","created_at":"2026-07-05T10:00:34Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.03618","created_at":"2026-07-05T10:00:34Z"},{"alias_kind":"pith_short_12","alias_value":"I6KL2I5IGOFN","created_at":"2026-07-05T10:00:34Z"},{"alias_kind":"pith_short_16","alias_value":"I6KL2I5IGOFNTCQF","created_at":"2026-07-05T10:00:34Z"},{"alias_kind":"pith_short_8","alias_value":"I6KL2I5I","created_at":"2026-07-05T10:00:34Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2023:I6KL2I5IGOFNTCQFAFBQO56LCZ","target":"record","payload":{"canonical_record":{"source":{"id":"2312.03618","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"math.OC","submitted_at":"2023-12-06T17:04:08Z","cross_cats_sorted":["cs.GT"],"title_canon_sha256":"8beb380635bb84adfeb60c207dd450f7f704b50a95869857272f0e731c421fce","abstract_canon_sha256":"12912bc1e11c6c68347d645a8f05221cf113a034fc2f61fbdc9d8b3fe1203f25"},"schema_version":"1.0"},"canonical_sha256":"4794bd23a8338ad98a0501430777cb166ae834799e71c1daec491351f50b76a0","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:00:34.016285Z","signature_b64":"tKj9dVmxJLN78NtU9MYP/nJS2n9WE5xNBtGpy56I563dGGPDOdAy1Pf0qp21MdVwNyBh6WCCPZQsZmjwTT9bBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4794bd23a8338ad98a0501430777cb166ae834799e71c1daec491351f50b76a0","last_reissued_at":"2026-07-05T10:00:34.015819Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:00:34.015819Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2312.03618","source_version":3,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:00:34Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"yVegUnUIoILAjOy6TrMr3tsNm0/7ojHJo1cfdtz1ro8BBp979DLAqx8tBaWgQaB/Hoqe88Ig4Jn7Rm4yxt1XBA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T20:51:27.205877Z"},"content_sha256":"573157b0d9de88cbdf4a0c28daf526277177dbcde283bf4f65db34a59b974460","schema_version":"1.0","event_id":"sha256:573157b0d9de88cbdf4a0c28daf526277177dbcde283bf4f65db34a59b974460"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2023:I6KL2I5IGOFNTCQFAFBQO56LCZ","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Beyond discounted returns: Robust Markov decision processes with average and Blackwell optimality","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.GT"],"primary_cat":"math.OC","authors_text":"Julien Grand-Cl\\'ement, Marek Petrik, Nicolas Vieille","submitted_at":"2023-12-06T17:04:08Z","abstract_excerpt":"Robust Markov Decision Processes (RMDPs) are a widely used framework for sequential decision-making under parameter uncertainty. RMDPs have been extensively studied when the objective is to maximize the discounted return, but little is known for average optimality (optimizing the long-run average of the rewards obtained over time) and Blackwell optimality (remaining discount optimal for all discount factors sufficiently close to ). In this paper, we prove several foundational results for RMDPs beyond the discounted return. We show that average optimal policies can be chosen stationary and dete"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.03618","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.03618/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:00:34Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"2784OUPl6hW71zUJATr7b1KqXnZS+MiqeAt5nzbhBAiIvAvY/gkfLW9A5VrHHxIxlqvkY46cacwCQeNPoE3hDw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T20:51:27.206376Z"},"content_sha256":"c42e801ea2c6515637795dde229138a202be387ee176505f528cff4ef48190a7","schema_version":"1.0","event_id":"sha256:c42e801ea2c6515637795dde229138a202be387ee176505f528cff4ef48190a7"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/I6KL2I5IGOFNTCQFAFBQO56LCZ/bundle.json","state_url":"https://pith.science/pith/I6KL2I5IGOFNTCQFAFBQO56LCZ/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/I6KL2I5IGOFNTCQFAFBQO56LCZ/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-09T20:51:27Z","links":{"resolver":"https://pith.science/pith/I6KL2I5IGOFNTCQFAFBQO56LCZ","bundle":"https://pith.science/pith/I6KL2I5IGOFNTCQFAFBQO56LCZ/bundle.json","state":"https://pith.science/pith/I6KL2I5IGOFNTCQFAFBQO56LCZ/state.json","well_known_bundle":"https://pith.science/.well-known/pith/I6KL2I5IGOFNTCQFAFBQO56LCZ/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2023:I6KL2I5IGOFNTCQFAFBQO56LCZ","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"12912bc1e11c6c68347d645a8f05221cf113a034fc2f61fbdc9d8b3fe1203f25","cross_cats_sorted":["cs.GT"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"math.OC","submitted_at":"2023-12-06T17:04:08Z","title_canon_sha256":"8beb380635bb84adfeb60c207dd450f7f704b50a95869857272f0e731c421fce"},"schema_version":"1.0","source":{"id":"2312.03618","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2312.03618","created_at":"2026-07-05T10:00:34Z"},{"alias_kind":"arxiv_version","alias_value":"2312.03618v3","created_at":"2026-07-05T10:00:34Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.03618","created_at":"2026-07-05T10:00:34Z"},{"alias_kind":"pith_short_12","alias_value":"I6KL2I5IGOFN","created_at":"2026-07-05T10:00:34Z"},{"alias_kind":"pith_short_16","alias_value":"I6KL2I5IGOFNTCQF","created_at":"2026-07-05T10:00:34Z"},{"alias_kind":"pith_short_8","alias_value":"I6KL2I5I","created_at":"2026-07-05T10:00:34Z"}],"graph_snapshots":[{"event_id":"sha256:c42e801ea2c6515637795dde229138a202be387ee176505f528cff4ef48190a7","target":"graph","created_at":"2026-07-05T10:00:34Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2312.03618/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Robust Markov Decision Processes (RMDPs) are a widely used framework for sequential decision-making under parameter uncertainty. RMDPs have been extensively studied when the objective is to maximize the discounted return, but little is known for average optimality (optimizing the long-run average of the rewards obtained over time) and Blackwell optimality (remaining discount optimal for all discount factors sufficiently close to ). In this paper, we prove several foundational results for RMDPs beyond the discounted return. We show that average optimal policies can be chosen stationary and dete","authors_text":"Julien Grand-Cl\\'ement, Marek Petrik, Nicolas Vieille","cross_cats":["cs.GT"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"math.OC","submitted_at":"2023-12-06T17:04:08Z","title":"Beyond discounted returns: Robust Markov decision processes with average and Blackwell optimality"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.03618","kind":"arxiv","version":3},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:573157b0d9de88cbdf4a0c28daf526277177dbcde283bf4f65db34a59b974460","target":"record","created_at":"2026-07-05T10:00:34Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"12912bc1e11c6c68347d645a8f05221cf113a034fc2f61fbdc9d8b3fe1203f25","cross_cats_sorted":["cs.GT"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"math.OC","submitted_at":"2023-12-06T17:04:08Z","title_canon_sha256":"8beb380635bb84adfeb60c207dd450f7f704b50a95869857272f0e731c421fce"},"schema_version":"1.0","source":{"id":"2312.03618","kind":"arxiv","version":3}},"canonical_sha256":"4794bd23a8338ad98a0501430777cb166ae834799e71c1daec491351f50b76a0","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"4794bd23a8338ad98a0501430777cb166ae834799e71c1daec491351f50b76a0","first_computed_at":"2026-07-05T10:00:34.015819Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T10:00:34.015819Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"tKj9dVmxJLN78NtU9MYP/nJS2n9WE5xNBtGpy56I563dGGPDOdAy1Pf0qp21MdVwNyBh6WCCPZQsZmjwTT9bBA==","signature_status":"signed_v1","signed_at":"2026-07-05T10:00:34.016285Z","signed_message":"canonical_sha256_bytes"},"source_id":"2312.03618","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:573157b0d9de88cbdf4a0c28daf526277177dbcde283bf4f65db34a59b974460","sha256:c42e801ea2c6515637795dde229138a202be387ee176505f528cff4ef48190a7"],"state_sha256":"2bb45bd7c25cdb94d6f6614ee113e0a8abed9556dbfbc6de158e4b577e6593aa"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"ZaDkktVPCl8kGAqxCrPD/fjHxU6qPc7iqAZqeOm+cProtJ3+d8yBodHXlHS58jf6eGk450G//FQtaMPY6NzSDA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-09T20:51:27.211393Z","bundle_sha256":"ff7f77652947fc34bdee4fced1c2d1df3045d60e171946b6011c95cfea8acfc8"}}