{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:NAYKQJD7QFEGGH5GYKW4NRHUCJ","short_pith_number":"pith:NAYKQJD7","canonical_record":{"source":{"id":"2403.15928","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2024-03-23T20:22:30Z","cross_cats_sorted":["math.OC"],"title_canon_sha256":"3c0fa58cf6c586c757a12fbe61b04dc4456982f14e264b82e91a3e3eba56a4cd","abstract_canon_sha256":"efc00b8cbf32b55c26e78c85491e9f28d40a1e1e3639f4be0fcf682db25bac70"},"schema_version":"1.0"},"canonical_sha256":"6830a8247f8148631fa6c2adc6c4f41262793646de99e7258980fad900e7d649","source":{"kind":"arxiv","id":"2403.15928","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2403.15928","created_at":"2026-07-05T08:00:08Z"},{"alias_kind":"arxiv_version","alias_value":"2403.15928v1","created_at":"2026-07-05T08:00:08Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.15928","created_at":"2026-07-05T08:00:08Z"},{"alias_kind":"pith_short_12","alias_value":"NAYKQJD7QFEG","created_at":"2026-07-05T08:00:08Z"},{"alias_kind":"pith_short_16","alias_value":"NAYKQJD7QFEGGH5G","created_at":"2026-07-05T08:00:08Z"},{"alias_kind":"pith_short_8","alias_value":"NAYKQJD7","created_at":"2026-07-05T08:00:08Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:NAYKQJD7QFEGGH5GYKW4NRHUCJ","target":"record","payload":{"canonical_record":{"source":{"id":"2403.15928","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2024-03-23T20:22:30Z","cross_cats_sorted":["math.OC"],"title_canon_sha256":"3c0fa58cf6c586c757a12fbe61b04dc4456982f14e264b82e91a3e3eba56a4cd","abstract_canon_sha256":"efc00b8cbf32b55c26e78c85491e9f28d40a1e1e3639f4be0fcf682db25bac70"},"schema_version":"1.0"},"canonical_sha256":"6830a8247f8148631fa6c2adc6c4f41262793646de99e7258980fad900e7d649","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:00:08.521946Z","signature_b64":"BYLrrkVvVQKf1zocAUD3Bcwv5DY6iG/Nxb/x/n11aDiNx8amW8JpZ7tVCxbBG+L9uF62xNJuKGhMIYUH2v/QDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6830a8247f8148631fa6c2adc6c4f41262793646de99e7258980fad900e7d649","last_reissued_at":"2026-07-05T08:00:08.521514Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:00:08.521514Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2403.15928","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:00:08Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"UpWWaEDxarv98mFqRb2s2oUtmG2omPFbYIanZ2xg+eMI7Mechs3sKuHfiwcBfnEZ2fEeVY8n3tVnos7EuyX4Bw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-21T17:21:16.200243Z"},"content_sha256":"598d001a1a60fc9a024d332ee2a8cd578f72b1319d540cda07a686d6324ea980","schema_version":"1.0","event_id":"sha256:598d001a1a60fc9a024d332ee2a8cd578f72b1319d540cda07a686d6324ea980"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:NAYKQJD7QFEGGH5GYKW4NRHUCJ","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Safe Reinforcement Learning for Constrained Markov Decision Processes with Stochastic Stopping Time","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["math.OC"],"primary_cat":"cs.LG","authors_text":"Abhijit Mazumdar, Manuela L. Bujorianu, Rafal Wisniewski","submitted_at":"2024-03-23T20:22:30Z","abstract_excerpt":"In this paper, we present an online reinforcement learning algorithm for constrained Markov decision processes with a safety constraint. Despite the necessary attention of the scientific community, considering stochastic stopping time, the problem of learning optimal policy without violating safety constraints during the learning phase is yet to be addressed. To this end, we propose an algorithm based on linear programming that does not require a process model. We show that the learned policy is safe with high confidence. We also propose a method to compute a safe baseline policy, which is cen"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.15928","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.15928/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:00:08Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"BvuqsH0SxRrUlLuz5HQphaZxq7jF9jx4SxyOP6lwXPJNmehTF3/v5UyPwTs1ygq7an0r6FjpvWvMy2uk5EKrBQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-21T17:21:16.200765Z"},"content_sha256":"7d8922dcdb5b98b1b85c26d85c59087a429bbeb3e8e9d8bcaea5b166c153b47e","schema_version":"1.0","event_id":"sha256:7d8922dcdb5b98b1b85c26d85c59087a429bbeb3e8e9d8bcaea5b166c153b47e"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/NAYKQJD7QFEGGH5GYKW4NRHUCJ/bundle.json","state_url":"https://pith.science/pith/NAYKQJD7QFEGGH5GYKW4NRHUCJ/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/NAYKQJD7QFEGGH5GYKW4NRHUCJ/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-21T17:21:16Z","links":{"resolver":"https://pith.science/pith/NAYKQJD7QFEGGH5GYKW4NRHUCJ","bundle":"https://pith.science/pith/NAYKQJD7QFEGGH5GYKW4NRHUCJ/bundle.json","state":"https://pith.science/pith/NAYKQJD7QFEGGH5GYKW4NRHUCJ/state.json","well_known_bundle":"https://pith.science/.well-known/pith/NAYKQJD7QFEGGH5GYKW4NRHUCJ/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:NAYKQJD7QFEGGH5GYKW4NRHUCJ","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"efc00b8cbf32b55c26e78c85491e9f28d40a1e1e3639f4be0fcf682db25bac70","cross_cats_sorted":["math.OC"],"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2024-03-23T20:22:30Z","title_canon_sha256":"3c0fa58cf6c586c757a12fbe61b04dc4456982f14e264b82e91a3e3eba56a4cd"},"schema_version":"1.0","source":{"id":"2403.15928","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2403.15928","created_at":"2026-07-05T08:00:08Z"},{"alias_kind":"arxiv_version","alias_value":"2403.15928v1","created_at":"2026-07-05T08:00:08Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.15928","created_at":"2026-07-05T08:00:08Z"},{"alias_kind":"pith_short_12","alias_value":"NAYKQJD7QFEG","created_at":"2026-07-05T08:00:08Z"},{"alias_kind":"pith_short_16","alias_value":"NAYKQJD7QFEGGH5G","created_at":"2026-07-05T08:00:08Z"},{"alias_kind":"pith_short_8","alias_value":"NAYKQJD7","created_at":"2026-07-05T08:00:08Z"}],"graph_snapshots":[{"event_id":"sha256:7d8922dcdb5b98b1b85c26d85c59087a429bbeb3e8e9d8bcaea5b166c153b47e","target":"graph","created_at":"2026-07-05T08:00:08Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2403.15928/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"In this paper, we present an online reinforcement learning algorithm for constrained Markov decision processes with a safety constraint. Despite the necessary attention of the scientific community, considering stochastic stopping time, the problem of learning optimal policy without violating safety constraints during the learning phase is yet to be addressed. To this end, we propose an algorithm based on linear programming that does not require a process model. We show that the learned policy is safe with high confidence. We also propose a method to compute a safe baseline policy, which is cen","authors_text":"Abhijit Mazumdar, Manuela L. Bujorianu, Rafal Wisniewski","cross_cats":["math.OC"],"headline":"","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2024-03-23T20:22:30Z","title":"Safe Reinforcement Learning for Constrained Markov Decision Processes with Stochastic Stopping Time"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.15928","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:598d001a1a60fc9a024d332ee2a8cd578f72b1319d540cda07a686d6324ea980","target":"record","created_at":"2026-07-05T08:00:08Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"efc00b8cbf32b55c26e78c85491e9f28d40a1e1e3639f4be0fcf682db25bac70","cross_cats_sorted":["math.OC"],"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2024-03-23T20:22:30Z","title_canon_sha256":"3c0fa58cf6c586c757a12fbe61b04dc4456982f14e264b82e91a3e3eba56a4cd"},"schema_version":"1.0","source":{"id":"2403.15928","kind":"arxiv","version":1}},"canonical_sha256":"6830a8247f8148631fa6c2adc6c4f41262793646de99e7258980fad900e7d649","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"6830a8247f8148631fa6c2adc6c4f41262793646de99e7258980fad900e7d649","first_computed_at":"2026-07-05T08:00:08.521514Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T08:00:08.521514Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"BYLrrkVvVQKf1zocAUD3Bcwv5DY6iG/Nxb/x/n11aDiNx8amW8JpZ7tVCxbBG+L9uF62xNJuKGhMIYUH2v/QDg==","signature_status":"signed_v1","signed_at":"2026-07-05T08:00:08.521946Z","signed_message":"canonical_sha256_bytes"},"source_id":"2403.15928","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:598d001a1a60fc9a024d332ee2a8cd578f72b1319d540cda07a686d6324ea980","sha256:7d8922dcdb5b98b1b85c26d85c59087a429bbeb3e8e9d8bcaea5b166c153b47e"],"state_sha256":"ec972bc7c2ce6acdfc3e75b11ccb213e3c37aaa46b171ca38c2bea811d46cc21"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"MCHt72I+pcsLz62RyvJ+e1ttWYpjxePkEfsijIliNXhtyd74/zUh8laNz2IjRlY/WxmX6mpd3ZgykGyTPeDHAA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-21T17:21:16.205649Z","bundle_sha256":"cd82dabb76e807cee33ed97d69255f1d62100ab531ab5d596abe95bee1ff64a7"}}