{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:MNVGPIK2M4XLJ7GOTYPB6TSRE2","short_pith_number":"pith:MNVGPIK2","schema_version":"1.0","canonical_sha256":"636a67a15a672eb4fcce9e1e1f4e51268f52a86522fc0d5803ae7228303cf732","source":{"kind":"arxiv","id":"2306.06265","version":1},"attestation_state":"computed","paper":{"title":"Near-optimal Conservative Exploration in Reinforcement Learning under Episode-wise Constraints","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.IT","math.IT","stat.ML"],"primary_cat":"cs.LG","authors_text":"Cong Shen, Donghao Li, Jing Yang, Ruiquan Huang","submitted_at":"2023-06-09T21:26:57Z","abstract_excerpt":"This paper investigates conservative exploration in reinforcement learning where the performance of the learning agent is guaranteed to be above a certain threshold throughout the learning process. It focuses on the tabular episodic Markov Decision Process (MDP) setting that has finite states and actions. With the knowledge of an existing safe baseline policy, an algorithm termed as StepMix is proposed to balance the exploitation and exploration while ensuring that the conservative constraint is never violated in each episode with high probability. StepMix features a unique design of a mixture"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.06265","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-06-09T21:26:57Z","cross_cats_sorted":["cs.IT","math.IT","stat.ML"],"title_canon_sha256":"53f87c2ea33825399707ada1a5a3a6676b73c3f32257bf337c26a2a215504753","abstract_canon_sha256":"38648b238bd12c29e3e984db92f260916cc96af51cb427ab9a88da9948e480ea"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:20:00.898077Z","signature_b64":"N90P7RMacreTHOa/9A95Mx2H5K2VvktwYXQ7sdOicgtvyadXiaMug5pTNBwoPUu3jzPtsmF7+Ox64iEUYRhpCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"636a67a15a672eb4fcce9e1e1f4e51268f52a86522fc0d5803ae7228303cf732","last_reissued_at":"2026-07-05T06:20:00.897638Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:20:00.897638Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Near-optimal Conservative Exploration in Reinforcement Learning under Episode-wise Constraints","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.IT","math.IT","stat.ML"],"primary_cat":"cs.LG","authors_text":"Cong Shen, Donghao Li, Jing Yang, Ruiquan Huang","submitted_at":"2023-06-09T21:26:57Z","abstract_excerpt":"This paper investigates conservative exploration in reinforcement learning where the performance of the learning agent is guaranteed to be above a certain threshold throughout the learning process. It focuses on the tabular episodic Markov Decision Process (MDP) setting that has finite states and actions. With the knowledge of an existing safe baseline policy, an algorithm termed as StepMix is proposed to balance the exploitation and exploration while ensuring that the conservative constraint is never violated in each episode with high probability. StepMix features a unique design of a mixture"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.06265","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.06265/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.06265","created_at":"2026-07-05T06:20:00.897696+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.06265v1","created_at":"2026-07-05T06:20:00.897696+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.06265","created_at":"2026-07-05T06:20:00.897696+00:00"},{"alias_kind":"pith_short_12","alias_value":"MNVGPIK2M4XL","created_at":"2026-07-05T06:20:00.897696+00:00"},{"alias_kind":"pith_short_16","alias_value":"MNVGPIK2M4XLJ7GO","created_at":"2026-07-05T06:20:00.897696+00:00"},{"alias_kind":"pith_short_8","alias_value":"MNVGPIK2","created_at":"2026-07-05T06:20:00.897696+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MNVGPIK2M4XLJ7GOTYPB6TSRE2","json":"https://pith.science/pith/MNVGPIK2M4XLJ7GOTYPB6TSRE2.json","graph_json":"https://pith.science/api/pith-number/MNVGPIK2M4XLJ7GOTYPB6TSRE2/graph.json","events_json":"https://pith.science/api/pith-number/MNVGPIK2M4XLJ7GOTYPB6TSRE2/events.json","paper":"https://pith.science/paper/MNVGPIK2"},"agent_actions":{"view_html":"https://pith.science/pith/MNVGPIK2M4XLJ7GOTYPB6TSRE2","download_json":"https://pith.science/pith/MNVGPIK2M4XLJ7GOTYPB6TSRE2.json","view_paper":"https://pith.science/paper/MNVGPIK2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.06265&json=true","fetch_graph":"https://pith.science/api/pith-number/MNVGPIK2M4XLJ7GOTYPB6TSRE2/graph.json","fetch_events":"https://pith.science/api/pith-number/MNVGPIK2M4XLJ7GOTYPB6TSRE2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MNVGPIK2M4XLJ7GOTYPB6TSRE2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MNVGPIK2M4XLJ7GOTYPB6TSRE2/action/storage_attestation","attest_author":"https://pith.science/pith/MNVGPIK2M4XLJ7GOTYPB6TSRE2/action/author_attestation","sign_citation":"https://pith.science/pith/MNVGPIK2M4XLJ7GOTYPB6TSRE2/action/citation_signature","submit_replication":"https://pith.science/pith/MNVGPIK2M4XLJ7GOTYPB6TSRE2/action/replication_record"}},"created_at":"2026-07-05T06:20:00.897696+00:00","updated_at":"2026-07-05T06:20:00.897696+00:00"}