{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2019:2C5FSIRA52J7TVFPG7ZXHOYFRS","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"0a8c4043e5c5eeab321d794b062cbae343b3a44b40ec54e2048f71b507a98896","cross_cats_sorted":["stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-08-27T21:53:42Z","title_canon_sha256":"8419ea6764de6f22c98234b7b2b7a6fe13f2b7dc825971612d0282471c1a85a7"},"schema_version":"1.0","source":{"id":"1908.10479","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1908.10479","created_at":"2026-07-05T00:00:17Z"},{"alias_kind":"arxiv_version","alias_value":"1908.10479v1","created_at":"2026-07-05T00:00:17Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1908.10479","created_at":"2026-07-05T00:00:17Z"},{"alias_kind":"pith_short_12","alias_value":"2C5FSIRA52J7","created_at":"2026-07-05T00:00:17Z"},{"alias_kind":"pith_short_16","alias_value":"2C5FSIRA52J7TVFP","created_at":"2026-07-05T00:00:17Z"},{"alias_kind":"pith_short_8","alias_value":"2C5FSIRA","created_at":"2026-07-05T00:00:17Z"}],"graph_snapshots":[{"event_id":"sha256:141addb2fd04b81308337258a0f334a28625feba402fde3a5c477149406b6574","target":"graph","created_at":"2026-07-05T00:00:17Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/1908.10479/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"We study algorithms for average-cost reinforcement learning problems with value function approximation. Our starting point is the recently proposed POLITEX algorithm, a version of policy iteration where the policy produced in each iteration is near-optimal in hindsight for the sum of all past value function estimates. POLITEX has sublinear regret guarantees in uniformly-mixing MDPs when the value estimation error can be controlled, which can be satisfied if all policies sufficiently explore the environment. Unfortunately, this assumption is often unrealistic. Motivated by the rapid growth of i","authors_text":"Csaba Szepesvari, Gellert Weisz, Nevena Lazic, Yasin Abbasi-Yadkori","cross_cats":["stat.ML"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-08-27T21:53:42Z","title":"Exploration-Enhanced POLITEX"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1908.10479","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:e6ead36af179ae8874cdc70788c0fa70d9386d61985afcf2612e7392c75bd111","target":"record","created_at":"2026-07-05T00:00:17Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"0a8c4043e5c5eeab321d794b062cbae343b3a44b40ec54e2048f71b507a98896","cross_cats_sorted":["stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-08-27T21:53:42Z","title_canon_sha256":"8419ea6764de6f22c98234b7b2b7a6fe13f2b7dc825971612d0282471c1a85a7"},"schema_version":"1.0","source":{"id":"1908.10479","kind":"arxiv","version":1}},"canonical_sha256":"d0ba592220ee93f9d4af37f373bb058c9b3df89ee2999840ff484bdf11d32fce","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"d0ba592220ee93f9d4af37f373bb058c9b3df89ee2999840ff484bdf11d32fce","first_computed_at":"2026-07-05T00:00:17.562373Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T00:00:17.562373Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"974KGQvasdt0sFg5tLQE3Tr3b4bHLXqfpkK4Rkb7/jyGI5va8YZ7UfNmyoY2gNzv47397SchUhznZAIsAm46Cw==","signature_status":"signed_v1","signed_at":"2026-07-05T00:00:17.562769Z","signed_message":"canonical_sha256_bytes"},"source_id":"1908.10479","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:e6ead36af179ae8874cdc70788c0fa70d9386d61985afcf2612e7392c75bd111","sha256:141addb2fd04b81308337258a0f334a28625feba402fde3a5c477149406b6574"],"state_sha256":"d76a14db4890cabdaab0cde672b0b7781a4423527f0a95cf38ba46c22d30c59a"}