{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:PY2R2XR7FZ6GC434VL6HBEELDD","short_pith_number":"pith:PY2R2XR7","schema_version":"1.0","canonical_sha256":"7e351d5e3f2e7c61737caafc70908b18d7e16e8ba006fdc4c5646e89ba2ec903","source":{"kind":"arxiv","id":"2311.05638","version":1},"attestation_state":"computed","paper":{"title":"Towards Instance-Optimality in Online PAC Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Andrea Tirinzoni, Aymen Al-Marjani, Emilie Kaufmann","submitted_at":"2023-10-31T19:26:36Z","abstract_excerpt":"Several recent works have proposed instance-dependent upper bounds on the number of episodes needed to identify, with probability $1-\\delta$, an $\\varepsilon$-optimal policy in finite-horizon tabular Markov Decision Processes (MDPs). These upper bounds feature various complexity measures for the MDP, which are defined based on different notions of sub-optimality gaps. However, as of now, no lower bound has been established to assess the optimality of any of these complexity measures, except for the special case of MDPs with deterministic transitions. In this paper, we propose the first instanc"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.05638","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"stat.ML","submitted_at":"2023-10-31T19:26:36Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"a6552b2cd9bb17e434b815951781285b3cdbfa4c5506e45c0eb1c205ed73cc19","abstract_canon_sha256":"78d95746c0d61a04b395b9047930980b1ab0195fd9fb1e969009b049d4e50812"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:11:20.669388Z","signature_b64":"jnR/Mfdx8m2DUilPFIWhtPHNO+1PUkTPnd/58rqU5UlzicsnN9zZVAEuUc6DHtb0bzhyia73TSOfmgT7llQzAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7e351d5e3f2e7c61737caafc70908b18d7e16e8ba006fdc4c5646e89ba2ec903","last_reissued_at":"2026-07-05T07:11:20.668928Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:11:20.668928Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards Instance-Optimality in Online PAC Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Andrea Tirinzoni, Aymen Al-Marjani, Emilie Kaufmann","submitted_at":"2023-10-31T19:26:36Z","abstract_excerpt":"Several recent works have proposed instance-dependent upper bounds on the number of episodes needed to identify, with probability $1-\\delta$, an $\\varepsilon$-optimal policy in finite-horizon tabular Markov Decision Processes (MDPs). These upper bounds feature various complexity measures for the MDP, which are defined based on different notions of sub-optimality gaps. However, as of now, no lower bound has been established to assess the optimality of any of these complexity measures, except for the special case of MDPs with deterministic transitions. In this paper, we propose the first instanc"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.05638","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.05638/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.05638","created_at":"2026-07-05T07:11:20.668984+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.05638v1","created_at":"2026-07-05T07:11:20.668984+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.05638","created_at":"2026-07-05T07:11:20.668984+00:00"},{"alias_kind":"pith_short_12","alias_value":"PY2R2XR7FZ6G","created_at":"2026-07-05T07:11:20.668984+00:00"},{"alias_kind":"pith_short_16","alias_value":"PY2R2XR7FZ6GC434","created_at":"2026-07-05T07:11:20.668984+00:00"},{"alias_kind":"pith_short_8","alias_value":"PY2R2XR7","created_at":"2026-07-05T07:11:20.668984+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.17515","citing_title":"Anytime-valid Optimal Policy Identification","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23182","citing_title":"Pure Exploration for a Good Policy in Reinforcement Learning with Bandit Feedback","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2502.03061","citing_title":"Pure Exploration Beyond Reward Feedback: The Role of Post-Action Context","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04979","citing_title":"On-line Learning in Tree MDPs by Treating Policies as Bandit Arms","ref_index":35,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PY2R2XR7FZ6GC434VL6HBEELDD","json":"https://pith.science/pith/PY2R2XR7FZ6GC434VL6HBEELDD.json","graph_json":"https://pith.science/api/pith-number/PY2R2XR7FZ6GC434VL6HBEELDD/graph.json","events_json":"https://pith.science/api/pith-number/PY2R2XR7FZ6GC434VL6HBEELDD/events.json","paper":"https://pith.science/paper/PY2R2XR7"},"agent_actions":{"view_html":"https://pith.science/pith/PY2R2XR7FZ6GC434VL6HBEELDD","download_json":"https://pith.science/pith/PY2R2XR7FZ6GC434VL6HBEELDD.json","view_paper":"https://pith.science/paper/PY2R2XR7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.05638&json=true","fetch_graph":"https://pith.science/api/pith-number/PY2R2XR7FZ6GC434VL6HBEELDD/graph.json","fetch_events":"https://pith.science/api/pith-number/PY2R2XR7FZ6GC434VL6HBEELDD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PY2R2XR7FZ6GC434VL6HBEELDD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PY2R2XR7FZ6GC434VL6HBEELDD/action/storage_attestation","attest_author":"https://pith.science/pith/PY2R2XR7FZ6GC434VL6HBEELDD/action/author_attestation","sign_citation":"https://pith.science/pith/PY2R2XR7FZ6GC434VL6HBEELDD/action/citation_signature","submit_replication":"https://pith.science/pith/PY2R2XR7FZ6GC434VL6HBEELDD/action/replication_record"}},"created_at":"2026-07-05T07:11:20.668984+00:00","updated_at":"2026-07-05T07:11:20.668984+00:00"}