{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2018:3FJ32ZMN2JSKWYWXX3KX4G2JA2","short_pith_number":"pith:3FJ32ZMN","canonical_record":{"source":{"id":"1805.11593","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2018-05-29T17:19:59Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"4e2b21fe03df05c11b7fd1275e6a7f6684795e44948427b2964532b568ce8f51","abstract_canon_sha256":"835fe03d93688f6eb8dd714c8c7a65704eb5c9b056173cf00af978280efcbd9f"},"schema_version":"1.0"},"canonical_sha256":"d953bd658dd264ab62d7bed57e1b490697d3f482a474334126f640f3aa9b96ca","source":{"kind":"arxiv","id":"1805.11593","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1805.11593","created_at":"2026-05-18T00:14:41Z"},{"alias_kind":"arxiv_version","alias_value":"1805.11593v1","created_at":"2026-05-18T00:14:41Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1805.11593","created_at":"2026-05-18T00:14:41Z"},{"alias_kind":"pith_short_12","alias_value":"3FJ32ZMN2JSK","created_at":"2026-05-18T12:32:02Z"},{"alias_kind":"pith_short_16","alias_value":"3FJ32ZMN2JSKWYWX","created_at":"2026-05-18T12:32:02Z"},{"alias_kind":"pith_short_8","alias_value":"3FJ32ZMN","created_at":"2026-05-18T12:32:02Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2018:3FJ32ZMN2JSKWYWXX3KX4G2JA2","target":"record","payload":{"canonical_record":{"source":{"id":"1805.11593","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2018-05-29T17:19:59Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"4e2b21fe03df05c11b7fd1275e6a7f6684795e44948427b2964532b568ce8f51","abstract_canon_sha256":"835fe03d93688f6eb8dd714c8c7a65704eb5c9b056173cf00af978280efcbd9f"},"schema_version":"1.0"},"canonical_sha256":"d953bd658dd264ab62d7bed57e1b490697d3f482a474334126f640f3aa9b96ca","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-18T00:14:41.932656Z","signature_b64":"LpVLA8KIvV++H0lhVFxOIjN4GZnHZ1sJNmdY2/G4PELpH9vz22CkUYtpC2NZtQ47V15YPXFRnADzOfX96YztDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d953bd658dd264ab62d7bed57e1b490697d3f482a474334126f640f3aa9b96ca","last_reissued_at":"2026-05-18T00:14:41.931888Z","signature_status":"signed_v1","first_computed_at":"2026-05-18T00:14:41.931888Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"1805.11593","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-18T00:14:41Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"svbWZ0BGBL94JAgCZN3CZWrJP/a9Usxv8M4ksrcKUfTq8MTdvDLwvEUjYb0LcSaJi3Bn0YN6t8AlKcP74a1IAQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-05-30T13:18:10.979533Z"},"content_sha256":"1fe68a98da199bfcc58324d549c7f63ae91d1fb165deae3823ac5441e15e6bfc","schema_version":"1.0","event_id":"sha256:1fe68a98da199bfcc58324d549c7f63ae91d1fb165deae3823ac5441e15e6bfc"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2018:3FJ32ZMN2JSKWYWXX3KX4G2JA2","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Observe and Look Further: Achieving Consistent Performance on Atari","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Bilal Piot, Dan Horgan, David Budden, Gabriel Barth-Maron, Hado van Hasselt, John Quan, Matteo Hessel, Mel Ve\\v{c}er\\'ik, Mohammad Gheshlaghi Azar, Olivier Pietquin, R\\'emi Munos, Tobias Pohlen, Todd Hester","submitted_at":"2018-05-29T17:19:59Z","abstract_excerpt":"Despite significant advances in the field of deep Reinforcement Learning (RL), today's algorithms still fail to learn human-level policies consistently over a set of diverse tasks such as Atari 2600 games. We identify three key challenges that any algorithm needs to master in order to perform well on all games: processing diverse reward distributions, reasoning over long time horizons, and exploring efficiently. In this paper, we propose an algorithm that addresses each of these challenges and is able to learn human-level policies on nearly all Atari games. A new transformed Bellman operator a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1805.11593","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-18T00:14:41Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"4NOKHolbOF1feFta6LJcGvMP396fjDRzCOFZacYqTkt8N/icV4XHzfVknQ78t5dikzwSp2d7euvM7HpJCZoeAw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-05-30T13:18:10.979886Z"},"content_sha256":"3d014dad045f6e37884c1b03c2d136def665e8a71c2ad8e7d95f43651af448c4","schema_version":"1.0","event_id":"sha256:3d014dad045f6e37884c1b03c2d136def665e8a71c2ad8e7d95f43651af448c4"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/3FJ32ZMN2JSKWYWXX3KX4G2JA2/bundle.json","state_url":"https://pith.science/pith/3FJ32ZMN2JSKWYWXX3KX4G2JA2/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/3FJ32ZMN2JSKWYWXX3KX4G2JA2/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-05-30T13:18:10Z","links":{"resolver":"https://pith.science/pith/3FJ32ZMN2JSKWYWXX3KX4G2JA2","bundle":"https://pith.science/pith/3FJ32ZMN2JSKWYWXX3KX4G2JA2/bundle.json","state":"https://pith.science/pith/3FJ32ZMN2JSKWYWXX3KX4G2JA2/state.json","well_known_bundle":"https://pith.science/.well-known/pith/3FJ32ZMN2JSKWYWXX3KX4G2JA2/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2018:3FJ32ZMN2JSKWYWXX3KX4G2JA2","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"835fe03d93688f6eb8dd714c8c7a65704eb5c9b056173cf00af978280efcbd9f","cross_cats_sorted":["cs.AI","stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2018-05-29T17:19:59Z","title_canon_sha256":"4e2b21fe03df05c11b7fd1275e6a7f6684795e44948427b2964532b568ce8f51"},"schema_version":"1.0","source":{"id":"1805.11593","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1805.11593","created_at":"2026-05-18T00:14:41Z"},{"alias_kind":"arxiv_version","alias_value":"1805.11593v1","created_at":"2026-05-18T00:14:41Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1805.11593","created_at":"2026-05-18T00:14:41Z"},{"alias_kind":"pith_short_12","alias_value":"3FJ32ZMN2JSK","created_at":"2026-05-18T12:32:02Z"},{"alias_kind":"pith_short_16","alias_value":"3FJ32ZMN2JSKWYWX","created_at":"2026-05-18T12:32:02Z"},{"alias_kind":"pith_short_8","alias_value":"3FJ32ZMN","created_at":"2026-05-18T12:32:02Z"}],"graph_snapshots":[{"event_id":"sha256:3d014dad045f6e37884c1b03c2d136def665e8a71c2ad8e7d95f43651af448c4","target":"graph","created_at":"2026-05-18T00:14:41Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"paper":{"abstract_excerpt":"Despite significant advances in the field of deep Reinforcement Learning (RL), today's algorithms still fail to learn human-level policies consistently over a set of diverse tasks such as Atari 2600 games. We identify three key challenges that any algorithm needs to master in order to perform well on all games: processing diverse reward distributions, reasoning over long time horizons, and exploring efficiently. In this paper, we propose an algorithm that addresses each of these challenges and is able to learn human-level policies on nearly all Atari games. A new transformed Bellman operator a","authors_text":"Bilal Piot, Dan Horgan, David Budden, Gabriel Barth-Maron, Hado van Hasselt, John Quan, Matteo Hessel, Mel Ve\\v{c}er\\'ik, Mohammad Gheshlaghi Azar, Olivier Pietquin, R\\'emi Munos, Tobias Pohlen, Todd Hester","cross_cats":["cs.AI","stat.ML"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2018-05-29T17:19:59Z","title":"Observe and Look Further: Achieving Consistent Performance on Atari"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1805.11593","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:1fe68a98da199bfcc58324d549c7f63ae91d1fb165deae3823ac5441e15e6bfc","target":"record","created_at":"2026-05-18T00:14:41Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"835fe03d93688f6eb8dd714c8c7a65704eb5c9b056173cf00af978280efcbd9f","cross_cats_sorted":["cs.AI","stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2018-05-29T17:19:59Z","title_canon_sha256":"4e2b21fe03df05c11b7fd1275e6a7f6684795e44948427b2964532b568ce8f51"},"schema_version":"1.0","source":{"id":"1805.11593","kind":"arxiv","version":1}},"canonical_sha256":"d953bd658dd264ab62d7bed57e1b490697d3f482a474334126f640f3aa9b96ca","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"d953bd658dd264ab62d7bed57e1b490697d3f482a474334126f640f3aa9b96ca","first_computed_at":"2026-05-18T00:14:41.931888Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-05-18T00:14:41.931888Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"LpVLA8KIvV++H0lhVFxOIjN4GZnHZ1sJNmdY2/G4PELpH9vz22CkUYtpC2NZtQ47V15YPXFRnADzOfX96YztDw==","signature_status":"signed_v1","signed_at":"2026-05-18T00:14:41.932656Z","signed_message":"canonical_sha256_bytes"},"source_id":"1805.11593","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:1fe68a98da199bfcc58324d549c7f63ae91d1fb165deae3823ac5441e15e6bfc","sha256:3d014dad045f6e37884c1b03c2d136def665e8a71c2ad8e7d95f43651af448c4"],"state_sha256":"1fc6505a6fb37fdcfdbc553ef72f84bf2326e2031df69425feb3c5369c434f2c"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"n4wkPbbKiB8/ZX1HhdXqwIKkZohjufitIFcREv/MkQHnWqYrZvt7dnLi2H7Yd171ku7qve8PkjTpa4btAScgAg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-05-30T13:18:10.982155Z","bundle_sha256":"36bb1fc0c22da43c250c951fe9972539897ab22a04b6b553de01a6bbaadb7b63"}}