{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2017:4KLO2QDFVB4VLWTDDIAOLMRLUM","short_pith_number":"pith:4KLO2QDF","canonical_record":{"source":{"id":"1710.06574","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2017-10-18T03:19:55Z","cross_cats_sorted":["cs.LG","stat.ML"],"title_canon_sha256":"9e2192a3bcfa168f7d0ca7bcd879b15bddcb516bb064d546b415599a585c3732","abstract_canon_sha256":"23642000dc6143544c4bb428157cb211238c0dc48db9756a13af07b9f55dbea5"},"schema_version":"1.0"},"canonical_sha256":"e296ed4065a87955da631a00e5b22ba33d70843caf280452a6abbe2d847f5c36","source":{"kind":"arxiv","id":"1710.06574","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1710.06574","created_at":"2026-05-18T00:32:33Z"},{"alias_kind":"arxiv_version","alias_value":"1710.06574v1","created_at":"2026-05-18T00:32:33Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1710.06574","created_at":"2026-05-18T00:32:33Z"},{"alias_kind":"pith_short_12","alias_value":"4KLO2QDFVB4V","created_at":"2026-05-18T12:31:00Z"},{"alias_kind":"pith_short_16","alias_value":"4KLO2QDFVB4VLWTD","created_at":"2026-05-18T12:31:00Z"},{"alias_kind":"pith_short_8","alias_value":"4KLO2QDF","created_at":"2026-05-18T12:31:00Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2017:4KLO2QDFVB4VLWTDDIAOLMRLUM","target":"record","payload":{"canonical_record":{"source":{"id":"1710.06574","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2017-10-18T03:19:55Z","cross_cats_sorted":["cs.LG","stat.ML"],"title_canon_sha256":"9e2192a3bcfa168f7d0ca7bcd879b15bddcb516bb064d546b415599a585c3732","abstract_canon_sha256":"23642000dc6143544c4bb428157cb211238c0dc48db9756a13af07b9f55dbea5"},"schema_version":"1.0"},"canonical_sha256":"e296ed4065a87955da631a00e5b22ba33d70843caf280452a6abbe2d847f5c36","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-18T00:32:33.338468Z","signature_b64":"Y9SrLcLFPrz2ZMv5Cx5joO5Lvz3fZeLAWOGcRJiMWuktqhtAyBC9kXhYBjYgqRN2qUnwDo0cCujRpxpel/+fAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e296ed4065a87955da631a00e5b22ba33d70843caf280452a6abbe2d847f5c36","last_reissued_at":"2026-05-18T00:32:33.337665Z","signature_status":"signed_v1","first_computed_at":"2026-05-18T00:32:33.337665Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"1710.06574","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-18T00:32:33Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"QylYmTC5AFd5f3nMWNXPTO2Fk0EZExFPCQe6ITpOrawG2cTkdExuacJ9fNJQZZfnDlbyDXNz9taqAW0bIfmyAQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-05-26T12:30:25.851475Z"},"content_sha256":"2c9abe8afbf1138d5458cd19911b07c45292dc039b66d6762677251ace857b11","schema_version":"1.0","event_id":"sha256:2c9abe8afbf1138d5458cd19911b07c45292dc039b66d6762677251ace857b11"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2017:4KLO2QDFVB4VLWTDDIAOLMRLUM","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"The Effects of Memory Replay in Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","stat.ML"],"primary_cat":"cs.AI","authors_text":"James Zou, Ruishan Liu","submitted_at":"2017-10-18T03:19:55Z","abstract_excerpt":"Experience replay is a key technique behind many recent advances in deep reinforcement learning. Allowing the agent to learn from earlier memories can speed up learning and break undesirable temporal correlations. Despite its wide-spread application, very little is understood about the properties of experience replay. How does the amount of memory kept affect learning dynamics? Does it help to prioritize certain experiences? In this paper, we address these questions by formulating a dynamical systems ODE model of Q-learning with experience replay. We derive analytic solutions of the ODE for a "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1710.06574","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-18T00:32:33Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"fnBzgnFJy4RAkXuIanHmKtmFsmcn26Uq+43irwvco5hePxFE8y0PJfBch05YprmvVrf9GZNxyVkt7M7KQ6ELDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-05-26T12:30:25.852175Z"},"content_sha256":"db1857745f2fdda42b0bb5eb8d05696cf43c9633e94fe8ba98abf3eefc21a7f1","schema_version":"1.0","event_id":"sha256:db1857745f2fdda42b0bb5eb8d05696cf43c9633e94fe8ba98abf3eefc21a7f1"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/4KLO2QDFVB4VLWTDDIAOLMRLUM/bundle.json","state_url":"https://pith.science/pith/4KLO2QDFVB4VLWTDDIAOLMRLUM/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/4KLO2QDFVB4VLWTDDIAOLMRLUM/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-05-26T12:30:25Z","links":{"resolver":"https://pith.science/pith/4KLO2QDFVB4VLWTDDIAOLMRLUM","bundle":"https://pith.science/pith/4KLO2QDFVB4VLWTDDIAOLMRLUM/bundle.json","state":"https://pith.science/pith/4KLO2QDFVB4VLWTDDIAOLMRLUM/state.json","well_known_bundle":"https://pith.science/.well-known/pith/4KLO2QDFVB4VLWTDDIAOLMRLUM/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2017:4KLO2QDFVB4VLWTDDIAOLMRLUM","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"23642000dc6143544c4bb428157cb211238c0dc48db9756a13af07b9f55dbea5","cross_cats_sorted":["cs.LG","stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2017-10-18T03:19:55Z","title_canon_sha256":"9e2192a3bcfa168f7d0ca7bcd879b15bddcb516bb064d546b415599a585c3732"},"schema_version":"1.0","source":{"id":"1710.06574","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1710.06574","created_at":"2026-05-18T00:32:33Z"},{"alias_kind":"arxiv_version","alias_value":"1710.06574v1","created_at":"2026-05-18T00:32:33Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1710.06574","created_at":"2026-05-18T00:32:33Z"},{"alias_kind":"pith_short_12","alias_value":"4KLO2QDFVB4V","created_at":"2026-05-18T12:31:00Z"},{"alias_kind":"pith_short_16","alias_value":"4KLO2QDFVB4VLWTD","created_at":"2026-05-18T12:31:00Z"},{"alias_kind":"pith_short_8","alias_value":"4KLO2QDF","created_at":"2026-05-18T12:31:00Z"}],"graph_snapshots":[{"event_id":"sha256:db1857745f2fdda42b0bb5eb8d05696cf43c9633e94fe8ba98abf3eefc21a7f1","target":"graph","created_at":"2026-05-18T00:32:33Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"paper":{"abstract_excerpt":"Experience replay is a key technique behind many recent advances in deep reinforcement learning. Allowing the agent to learn from earlier memories can speed up learning and break undesirable temporal correlations. Despite its wide-spread application, very little is understood about the properties of experience replay. How does the amount of memory kept affect learning dynamics? Does it help to prioritize certain experiences? In this paper, we address these questions by formulating a dynamical systems ODE model of Q-learning with experience replay. We derive analytic solutions of the ODE for a ","authors_text":"James Zou, Ruishan Liu","cross_cats":["cs.LG","stat.ML"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2017-10-18T03:19:55Z","title":"The Effects of Memory Replay in Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1710.06574","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:2c9abe8afbf1138d5458cd19911b07c45292dc039b66d6762677251ace857b11","target":"record","created_at":"2026-05-18T00:32:33Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"23642000dc6143544c4bb428157cb211238c0dc48db9756a13af07b9f55dbea5","cross_cats_sorted":["cs.LG","stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2017-10-18T03:19:55Z","title_canon_sha256":"9e2192a3bcfa168f7d0ca7bcd879b15bddcb516bb064d546b415599a585c3732"},"schema_version":"1.0","source":{"id":"1710.06574","kind":"arxiv","version":1}},"canonical_sha256":"e296ed4065a87955da631a00e5b22ba33d70843caf280452a6abbe2d847f5c36","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"e296ed4065a87955da631a00e5b22ba33d70843caf280452a6abbe2d847f5c36","first_computed_at":"2026-05-18T00:32:33.337665Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-05-18T00:32:33.337665Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"Y9SrLcLFPrz2ZMv5Cx5joO5Lvz3fZeLAWOGcRJiMWuktqhtAyBC9kXhYBjYgqRN2qUnwDo0cCujRpxpel/+fAA==","signature_status":"signed_v1","signed_at":"2026-05-18T00:32:33.338468Z","signed_message":"canonical_sha256_bytes"},"source_id":"1710.06574","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:2c9abe8afbf1138d5458cd19911b07c45292dc039b66d6762677251ace857b11","sha256:db1857745f2fdda42b0bb5eb8d05696cf43c9633e94fe8ba98abf3eefc21a7f1"],"state_sha256":"c590089bc624701669f5c05e45a2373c143fe22971e4b103866d2830b374b18c"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"/mGu9WN5isalNa55cBTURQz6vwG44hX6w4chscNQH+UxAYdoEFWSmMNIGRyXIQHFo90l8Tmhkf32HxEO9v3TDw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-05-26T12:30:25.855239Z","bundle_sha256":"62fe17f8bc1a40acc8622531bd7f457ecafa75d1ee36687765634461587f9d7a"}}