{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:CYO5WUDZJZTBJAIWKSBZP32QEW","short_pith_number":"pith:CYO5WUDZ","schema_version":"1.0","canonical_sha256":"161ddb50794e66148116548397ef5025aad4f0721ffdf1e6e061ac39c563d663","source":{"kind":"arxiv","id":"2301.11547","version":1},"attestation_state":"computed","paper":{"title":"Safe Posterior Sampling for Constrained MDPs with Bounded Constraint Violation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.SY","eess.SY"],"primary_cat":"cs.LG","authors_text":"Krishna C Kalagarla, Pierluigi Nuzzo, Rahul Jain","submitted_at":"2023-01-27T06:18:25Z","abstract_excerpt":"Constrained Markov decision processes (CMDPs) model scenarios of sequential decision making with multiple objectives that are increasingly important in many applications. However, the model is often unknown and must be learned online while still ensuring the constraint is met, or at least the violation is bounded with time. Some recent papers have made progress on this very challenging problem but either need unsatisfactory assumptions such as knowledge of a safe policy, or have high cumulative regret. We propose the Safe PSRL (posterior sampling-based RL) algorithm that does not need such ass"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2301.11547","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-01-27T06:18:25Z","cross_cats_sorted":["cs.AI","cs.SY","eess.SY"],"title_canon_sha256":"5d98bb9e295751209dc5953b4d1a3aef689b7c9658cca170eb8b5cd37070ca23","abstract_canon_sha256":"f6b7fcd8c91e6d6bca1d8ae86e7a6601cddefe1dd5d9eddd727ab16f0c37b3e9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:36:23.424651Z","signature_b64":"ynBWh3++SYnk1wxyJH3ACVYlkGu+/n2W3jjZ+K/59osl64eBaSB032vXAJJDO3TEScJWV2Q5OljQKGi+nnOFDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"161ddb50794e66148116548397ef5025aad4f0721ffdf1e6e061ac39c563d663","last_reissued_at":"2026-07-05T05:36:23.424153Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:36:23.424153Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Safe Posterior Sampling for Constrained MDPs with Bounded Constraint Violation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.SY","eess.SY"],"primary_cat":"cs.LG","authors_text":"Krishna C Kalagarla, Pierluigi Nuzzo, Rahul Jain","submitted_at":"2023-01-27T06:18:25Z","abstract_excerpt":"Constrained Markov decision processes (CMDPs) model scenarios of sequential decision making with multiple objectives that are increasingly important in many applications. However, the model is often unknown and must be learned online while still ensuring the constraint is met, or at least the violation is bounded with time. Some recent papers have made progress on this very challenging problem but either need unsatisfactory assumptions such as knowledge of a safe policy, or have high cumulative regret. We propose the Safe PSRL (posterior sampling-based RL) algorithm that does not need such ass"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2301.11547","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2301.11547/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2301.11547","created_at":"2026-07-05T05:36:23.424211+00:00"},{"alias_kind":"arxiv_version","alias_value":"2301.11547v1","created_at":"2026-07-05T05:36:23.424211+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2301.11547","created_at":"2026-07-05T05:36:23.424211+00:00"},{"alias_kind":"pith_short_12","alias_value":"CYO5WUDZJZTB","created_at":"2026-07-05T05:36:23.424211+00:00"},{"alias_kind":"pith_short_16","alias_value":"CYO5WUDZJZTBJAIW","created_at":"2026-07-05T05:36:23.424211+00:00"},{"alias_kind":"pith_short_8","alias_value":"CYO5WUDZ","created_at":"2026-07-05T05:36:23.424211+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CYO5WUDZJZTBJAIWKSBZP32QEW","json":"https://pith.science/pith/CYO5WUDZJZTBJAIWKSBZP32QEW.json","graph_json":"https://pith.science/api/pith-number/CYO5WUDZJZTBJAIWKSBZP32QEW/graph.json","events_json":"https://pith.science/api/pith-number/CYO5WUDZJZTBJAIWKSBZP32QEW/events.json","paper":"https://pith.science/paper/CYO5WUDZ"},"agent_actions":{"view_html":"https://pith.science/pith/CYO5WUDZJZTBJAIWKSBZP32QEW","download_json":"https://pith.science/pith/CYO5WUDZJZTBJAIWKSBZP32QEW.json","view_paper":"https://pith.science/paper/CYO5WUDZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2301.11547&json=true","fetch_graph":"https://pith.science/api/pith-number/CYO5WUDZJZTBJAIWKSBZP32QEW/graph.json","fetch_events":"https://pith.science/api/pith-number/CYO5WUDZJZTBJAIWKSBZP32QEW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CYO5WUDZJZTBJAIWKSBZP32QEW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CYO5WUDZJZTBJAIWKSBZP32QEW/action/storage_attestation","attest_author":"https://pith.science/pith/CYO5WUDZJZTBJAIWKSBZP32QEW/action/author_attestation","sign_citation":"https://pith.science/pith/CYO5WUDZJZTBJAIWKSBZP32QEW/action/citation_signature","submit_replication":"https://pith.science/pith/CYO5WUDZJZTBJAIWKSBZP32QEW/action/replication_record"}},"created_at":"2026-07-05T05:36:23.424211+00:00","updated_at":"2026-07-05T05:36:23.424211+00:00"}