{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2019:WFFV5RGVBV2ITJYG7KHAWO25RF","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"f3a64e712e0353f6e6d24a74959573b1fb1b7412143eb4252a5b1ed97ae0f6c3","cross_cats_sorted":["stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-08-06T22:36:35Z","title_canon_sha256":"8b8f55d828d9b9454972d12a47a6c0460f3c8a982a07f1b68500fadb9391c1ff"},"schema_version":"1.0","source":{"id":"1908.02388","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1908.02388","created_at":"2026-07-05T03:17:11Z"},{"alias_kind":"arxiv_version","alias_value":"1908.02388v3","created_at":"2026-07-05T03:17:11Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1908.02388","created_at":"2026-07-05T03:17:11Z"},{"alias_kind":"pith_short_12","alias_value":"WFFV5RGVBV2I","created_at":"2026-07-05T03:17:11Z"},{"alias_kind":"pith_short_16","alias_value":"WFFV5RGVBV2ITJYG","created_at":"2026-07-05T03:17:11Z"},{"alias_kind":"pith_short_8","alias_value":"WFFV5RGV","created_at":"2026-07-05T03:17:11Z"}],"graph_snapshots":[{"event_id":"sha256:a055b85691a1225862b128da45c640fa28abb5360857ed47c7c2e702a280ed46","target":"graph","created_at":"2026-07-05T03:17:11Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/1908.02388/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"This paper provides an empirical evaluation of recently developed exploration algorithms within the Arcade Learning Environment (ALE). We study the use of different reward bonuses that incentives exploration in reinforcement learning. We do so by fixing the learning algorithm used and focusing only on the impact of the different exploration bonuses in the agent's performance. We use Rainbow, the state-of-the-art algorithm for value-based agents, and focus on some of the bonuses proposed in the last few years. We consider the impact these algorithms have on performance within the popular game M","authors_text":"Aaron Courville, Adrien Ali Ta\\\"iga, Marc G. Bellemare, Marlos C. Machado, William Fedus","cross_cats":["stat.ML"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-08-06T22:36:35Z","title":"Benchmarking Bonus-Based Exploration Methods on the Arcade Learning Environment"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1908.02388","kind":"arxiv","version":3},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:8d0e762bfb5b2511a29eb4395d0ee9c17a0d126d312d842cebabee0da2323cbd","target":"record","created_at":"2026-07-05T03:17:11Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"f3a64e712e0353f6e6d24a74959573b1fb1b7412143eb4252a5b1ed97ae0f6c3","cross_cats_sorted":["stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-08-06T22:36:35Z","title_canon_sha256":"8b8f55d828d9b9454972d12a47a6c0460f3c8a982a07f1b68500fadb9391c1ff"},"schema_version":"1.0","source":{"id":"1908.02388","kind":"arxiv","version":3}},"canonical_sha256":"b14b5ec4d50d7489a706fa8e0b3b5d895badc4a72a3f12c432999acca2a08e64","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"b14b5ec4d50d7489a706fa8e0b3b5d895badc4a72a3f12c432999acca2a08e64","first_computed_at":"2026-07-05T03:17:11.725646Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T03:17:11.725646Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"twHNoAHA1NVCheDBgF3zPjaO762NlfpCtYvlpfOok7jc13BOS5PWQ/qX3ptpbYCFGuLRDH3RMATKH94qgjYZDA==","signature_status":"signed_v1","signed_at":"2026-07-05T03:17:11.726179Z","signed_message":"canonical_sha256_bytes"},"source_id":"1908.02388","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:8d0e762bfb5b2511a29eb4395d0ee9c17a0d126d312d842cebabee0da2323cbd","sha256:a055b85691a1225862b128da45c640fa28abb5360857ed47c7c2e702a280ed46"],"state_sha256":"f6ed75592f52d0ade1caaca06db2c5990b747784eca6cd14b0c95b8e75983995"}