{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:TXECSPNK7JBRPHD5WZ66U4VUIK","short_pith_number":"pith:TXECSPNK","schema_version":"1.0","canonical_sha256":"9dc8293daafa43179c7db67dea72b442a6730945be5e35efc88b1050dc289692","source":{"kind":"arxiv","id":"1910.12807","version":1},"attestation_state":"computed","paper":{"title":"Better Exploration with Optimistic Actor-Critic","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Kamil Ciosek, Katja Hofmann, Quan Vuong, Robert Loftin","submitted_at":"2019-10-28T17:06:40Z","abstract_excerpt":"Actor-critic methods, a type of model-free Reinforcement Learning, have been successfully applied to challenging tasks in continuous control, often achieving state-of-the art performance. However, wide-scale adoption of these methods in real-world domains is made difficult by their poor sample efficiency. We address this problem both theoretically and empirically. On the theoretical side, we identify two phenomena preventing efficient exploration in existing state-of-the-art algorithms such as Soft Actor Critic. First, combining a greedy actor update with a pessimistic estimate of the critic l"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1910.12807","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"stat.ML","submitted_at":"2019-10-28T17:06:40Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"7aa410885e36eb7eb3200b051e789e732e93d6d8d0ca0118f0e3846f2ea13f79","abstract_canon_sha256":"9131fab4a27fcc8c713d4a401b9cc6952f0d4271b34efe04a5c87d2debfe18c3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:15:20.248400Z","signature_b64":"XhmLnHbHLN8XlRiQWaXqlf3/yXiORq9k4ao/NlUfrHgd8R7wkbL73+Ke5sxqkQQP+h7H3idz54lUlRYcbmX0Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9dc8293daafa43179c7db67dea72b442a6730945be5e35efc88b1050dc289692","last_reissued_at":"2026-07-05T00:15:20.248005Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:15:20.248005Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Better Exploration with Optimistic Actor-Critic","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Kamil Ciosek, Katja Hofmann, Quan Vuong, Robert Loftin","submitted_at":"2019-10-28T17:06:40Z","abstract_excerpt":"Actor-critic methods, a type of model-free Reinforcement Learning, have been successfully applied to challenging tasks in continuous control, often achieving state-of-the art performance. However, wide-scale adoption of these methods in real-world domains is made difficult by their poor sample efficiency. We address this problem both theoretically and empirically. On the theoretical side, we identify two phenomena preventing efficient exploration in existing state-of-the-art algorithms such as Soft Actor Critic. First, combining a greedy actor update with a pessimistic estimate of the critic l"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1910.12807","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1910.12807/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1910.12807","created_at":"2026-07-05T00:15:20.248062+00:00"},{"alias_kind":"arxiv_version","alias_value":"1910.12807v1","created_at":"2026-07-05T00:15:20.248062+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1910.12807","created_at":"2026-07-05T00:15:20.248062+00:00"},{"alias_kind":"pith_short_12","alias_value":"TXECSPNK7JBR","created_at":"2026-07-05T00:15:20.248062+00:00"},{"alias_kind":"pith_short_16","alias_value":"TXECSPNK7JBRPHD5","created_at":"2026-07-05T00:15:20.248062+00:00"},{"alias_kind":"pith_short_8","alias_value":"TXECSPNK","created_at":"2026-07-05T00:15:20.248062+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TXECSPNK7JBRPHD5WZ66U4VUIK","json":"https://pith.science/pith/TXECSPNK7JBRPHD5WZ66U4VUIK.json","graph_json":"https://pith.science/api/pith-number/TXECSPNK7JBRPHD5WZ66U4VUIK/graph.json","events_json":"https://pith.science/api/pith-number/TXECSPNK7JBRPHD5WZ66U4VUIK/events.json","paper":"https://pith.science/paper/TXECSPNK"},"agent_actions":{"view_html":"https://pith.science/pith/TXECSPNK7JBRPHD5WZ66U4VUIK","download_json":"https://pith.science/pith/TXECSPNK7JBRPHD5WZ66U4VUIK.json","view_paper":"https://pith.science/paper/TXECSPNK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1910.12807&json=true","fetch_graph":"https://pith.science/api/pith-number/TXECSPNK7JBRPHD5WZ66U4VUIK/graph.json","fetch_events":"https://pith.science/api/pith-number/TXECSPNK7JBRPHD5WZ66U4VUIK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TXECSPNK7JBRPHD5WZ66U4VUIK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TXECSPNK7JBRPHD5WZ66U4VUIK/action/storage_attestation","attest_author":"https://pith.science/pith/TXECSPNK7JBRPHD5WZ66U4VUIK/action/author_attestation","sign_citation":"https://pith.science/pith/TXECSPNK7JBRPHD5WZ66U4VUIK/action/citation_signature","submit_replication":"https://pith.science/pith/TXECSPNK7JBRPHD5WZ66U4VUIK/action/replication_record"}},"created_at":"2026-07-05T00:15:20.248062+00:00","updated_at":"2026-07-05T00:15:20.248062+00:00"}