{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:GURYHXTJM6DOALTJDENKZLCGSE","short_pith_number":"pith:GURYHXTJ","schema_version":"1.0","canonical_sha256":"352383de696786e02e69191aacac469109db4e7bc0b93d7e067ad806fb019d1b","source":{"kind":"arxiv","id":"2302.09339","version":2},"attestation_state":"computed","paper":{"title":"Efficient Exploration via Epistemic-Risk-Seeking Policy Optimization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Brendan O'Donoghue","submitted_at":"2023-02-18T14:13:25Z","abstract_excerpt":"Exploration remains a key challenge in deep reinforcement learning (RL). Optimism in the face of uncertainty is a well-known heuristic with theoretical guarantees in the tabular setting, but how best to translate the principle to deep reinforcement learning, which involves online stochastic gradients and deep network function approximators, is not fully understood. In this paper we propose a new, differentiable optimistic objective that when optimized yields a policy that provably explores efficiently, with guarantees even under function approximation. Our new objective is a zero-sum two-playe"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2302.09339","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-02-18T14:13:25Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"a4f0afa0c7ce9466ee1d603f81622a716616cfb31ee1350c76c374411524bf21","abstract_canon_sha256":"73d4db6b04d83bf172ad38a5955f3da961a229abe652a5940547a6f71fa00b18"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:17:04.966216Z","signature_b64":"DZB/6I7DDtHrBAqfz8w4iGzJZSHTC6WM3trHOxvIz7gaDT49zWu4R8rx94OQNgRV5+4+28aJIrG6DuB9i5/vAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"352383de696786e02e69191aacac469109db4e7bc0b93d7e067ad806fb019d1b","last_reissued_at":"2026-07-05T06:17:04.965761Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:17:04.965761Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Efficient Exploration via Epistemic-Risk-Seeking Policy Optimization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Brendan O'Donoghue","submitted_at":"2023-02-18T14:13:25Z","abstract_excerpt":"Exploration remains a key challenge in deep reinforcement learning (RL). Optimism in the face of uncertainty is a well-known heuristic with theoretical guarantees in the tabular setting, but how best to translate the principle to deep reinforcement learning, which involves online stochastic gradients and deep network function approximators, is not fully understood. In this paper we propose a new, differentiable optimistic objective that when optimized yields a policy that provably explores efficiently, with guarantees even under function approximation. Our new objective is a zero-sum two-playe"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2302.09339","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2302.09339/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2302.09339","created_at":"2026-07-05T06:17:04.965812+00:00"},{"alias_kind":"arxiv_version","alias_value":"2302.09339v2","created_at":"2026-07-05T06:17:04.965812+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2302.09339","created_at":"2026-07-05T06:17:04.965812+00:00"},{"alias_kind":"pith_short_12","alias_value":"GURYHXTJM6DO","created_at":"2026-07-05T06:17:04.965812+00:00"},{"alias_kind":"pith_short_16","alias_value":"GURYHXTJM6DOALTJ","created_at":"2026-07-05T06:17:04.965812+00:00"},{"alias_kind":"pith_short_8","alias_value":"GURYHXTJ","created_at":"2026-07-05T06:17:04.965812+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2602.06239","citing_title":"Provably avoiding over-optimization in Direct Preference Optimization without knowing the data distribution","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2602.06239","citing_title":"Provably avoiding over-optimization in Direct Preference Optimization without knowing the data distribution","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GURYHXTJM6DOALTJDENKZLCGSE","json":"https://pith.science/pith/GURYHXTJM6DOALTJDENKZLCGSE.json","graph_json":"https://pith.science/api/pith-number/GURYHXTJM6DOALTJDENKZLCGSE/graph.json","events_json":"https://pith.science/api/pith-number/GURYHXTJM6DOALTJDENKZLCGSE/events.json","paper":"https://pith.science/paper/GURYHXTJ"},"agent_actions":{"view_html":"https://pith.science/pith/GURYHXTJM6DOALTJDENKZLCGSE","download_json":"https://pith.science/pith/GURYHXTJM6DOALTJDENKZLCGSE.json","view_paper":"https://pith.science/paper/GURYHXTJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2302.09339&json=true","fetch_graph":"https://pith.science/api/pith-number/GURYHXTJM6DOALTJDENKZLCGSE/graph.json","fetch_events":"https://pith.science/api/pith-number/GURYHXTJM6DOALTJDENKZLCGSE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GURYHXTJM6DOALTJDENKZLCGSE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GURYHXTJM6DOALTJDENKZLCGSE/action/storage_attestation","attest_author":"https://pith.science/pith/GURYHXTJM6DOALTJDENKZLCGSE/action/author_attestation","sign_citation":"https://pith.science/pith/GURYHXTJM6DOALTJDENKZLCGSE/action/citation_signature","submit_replication":"https://pith.science/pith/GURYHXTJM6DOALTJDENKZLCGSE/action/replication_record"}},"created_at":"2026-07-05T06:17:04.965812+00:00","updated_at":"2026-07-05T06:17:04.965812+00:00"}