{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2021:ZQDD7XCTGKHGU3GI7HT3V75LJL","short_pith_number":"pith:ZQDD7XCT","canonical_record":{"source":{"id":"2105.07253","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-05-15T16:08:45Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"57e7cf3ef10d2bf8517470e636bff10f829303bed7ab72226d031c8e86687c11","abstract_canon_sha256":"d04d4a0d2df4dc38858de2ff850b284786fbb154d48776022b3ca37e1bfdc000"},"schema_version":"1.0"},"canonical_sha256":"cc063fdc53328e6a6cc8f9e7baffab4ad0119fafaa5f4f1f098a4c17ca14517b","source":{"kind":"arxiv","id":"2105.07253","version":3},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2105.07253","created_at":"2026-07-05T03:30:11Z"},{"alias_kind":"arxiv_version","alias_value":"2105.07253v3","created_at":"2026-07-05T03:30:11Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2105.07253","created_at":"2026-07-05T03:30:11Z"},{"alias_kind":"pith_short_12","alias_value":"ZQDD7XCTGKHG","created_at":"2026-07-05T03:30:11Z"},{"alias_kind":"pith_short_16","alias_value":"ZQDD7XCTGKHGU3GI","created_at":"2026-07-05T03:30:11Z"},{"alias_kind":"pith_short_8","alias_value":"ZQDD7XCT","created_at":"2026-07-05T03:30:11Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2021:ZQDD7XCTGKHGU3GI7HT3V75LJL","target":"record","payload":{"canonical_record":{"source":{"id":"2105.07253","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-05-15T16:08:45Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"57e7cf3ef10d2bf8517470e636bff10f829303bed7ab72226d031c8e86687c11","abstract_canon_sha256":"d04d4a0d2df4dc38858de2ff850b284786fbb154d48776022b3ca37e1bfdc000"},"schema_version":"1.0"},"canonical_sha256":"cc063fdc53328e6a6cc8f9e7baffab4ad0119fafaa5f4f1f098a4c17ca14517b","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:30:11.735968Z","signature_b64":"ld7Qk1nGmNqs9jUAZszlIuplsf6F9hHUB+uehOLr048K0vDDzzZu9+4vOzzoMjV6Q/Git1fXIETURrE5zZeLAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cc063fdc53328e6a6cc8f9e7baffab4ad0119fafaa5f4f1f098a4c17ca14517b","last_reissued_at":"2026-07-05T03:30:11.735566Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:30:11.735566Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2105.07253","source_version":3,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T03:30:11Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"s6RkhijqUinozsiAzWujSMMH9qYmbgjaz66b2sKnAXYSIxpyWinfh4fv4Kn11wgEIrZqo7lzb5BLiMdcZ2q+Dw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-07-24T19:25:13.948801Z"},"content_sha256":"5dd10dc6ab006c879839287a03442d4cb9e49a6541f3777fcfc1c7508e28785a","schema_version":"1.0","event_id":"sha256:5dd10dc6ab006c879839287a03442d4cb9e49a6541f3777fcfc1c7508e28785a"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2021:ZQDD7XCTGKHGU3GI7HT3V75LJL","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Regret Minimization Experience Replay in Off-Policy Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Feng Xu, Jing-cheng Pang, Shengyi Jiang, Xu-Hui Liu, Yang Yu, Zhenghai Xue","submitted_at":"2021-05-15T16:08:45Z","abstract_excerpt":"In reinforcement learning, experience replay stores past samples for further reuse. Prioritized sampling is a promising technique to better utilize these samples. Previous criteria of prioritization include TD error, recentness and corrective feedback, which are mostly heuristically designed. In this work, we start from the regret minimization objective, and obtain an optimal prioritization strategy for Bellman update that can directly maximize the return of the policy. The theory suggests that data with higher hindsight TD error, better on-policiness and more accurate Q value should be assign"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2105.07253","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2105.07253/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T03:30:11Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Dwt8xrvEFmOgrcderhjNTU0zbBpmaFvUUxsYMlPy4eEA1JkaL/8mqGRbzhcl5mYhrY3a0CUTqOKRca/zv53kDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-07-24T19:25:13.949169Z"},"content_sha256":"e0a0c9e8f61a5ff9853dbe803981c60009d2066de556bd40ab297c86e66567ea","schema_version":"1.0","event_id":"sha256:e0a0c9e8f61a5ff9853dbe803981c60009d2066de556bd40ab297c86e66567ea"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/ZQDD7XCTGKHGU3GI7HT3V75LJL/bundle.json","state_url":"https://pith.science/pith/ZQDD7XCTGKHGU3GI7HT3V75LJL/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/ZQDD7XCTGKHGU3GI7HT3V75LJL/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-07-24T19:25:13Z","links":{"resolver":"https://pith.science/pith/ZQDD7XCTGKHGU3GI7HT3V75LJL","bundle":"https://pith.science/pith/ZQDD7XCTGKHGU3GI7HT3V75LJL/bundle.json","state":"https://pith.science/pith/ZQDD7XCTGKHGU3GI7HT3V75LJL/state.json","well_known_bundle":"https://pith.science/.well-known/pith/ZQDD7XCTGKHGU3GI7HT3V75LJL/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2021:ZQDD7XCTGKHGU3GI7HT3V75LJL","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"d04d4a0d2df4dc38858de2ff850b284786fbb154d48776022b3ca37e1bfdc000","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-05-15T16:08:45Z","title_canon_sha256":"57e7cf3ef10d2bf8517470e636bff10f829303bed7ab72226d031c8e86687c11"},"schema_version":"1.0","source":{"id":"2105.07253","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2105.07253","created_at":"2026-07-05T03:30:11Z"},{"alias_kind":"arxiv_version","alias_value":"2105.07253v3","created_at":"2026-07-05T03:30:11Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2105.07253","created_at":"2026-07-05T03:30:11Z"},{"alias_kind":"pith_short_12","alias_value":"ZQDD7XCTGKHG","created_at":"2026-07-05T03:30:11Z"},{"alias_kind":"pith_short_16","alias_value":"ZQDD7XCTGKHGU3GI","created_at":"2026-07-05T03:30:11Z"},{"alias_kind":"pith_short_8","alias_value":"ZQDD7XCT","created_at":"2026-07-05T03:30:11Z"}],"graph_snapshots":[{"event_id":"sha256:e0a0c9e8f61a5ff9853dbe803981c60009d2066de556bd40ab297c86e66567ea","target":"graph","created_at":"2026-07-05T03:30:11Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2105.07253/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"In reinforcement learning, experience replay stores past samples for further reuse. Prioritized sampling is a promising technique to better utilize these samples. Previous criteria of prioritization include TD error, recentness and corrective feedback, which are mostly heuristically designed. In this work, we start from the regret minimization objective, and obtain an optimal prioritization strategy for Bellman update that can directly maximize the return of the policy. The theory suggests that data with higher hindsight TD error, better on-policiness and more accurate Q value should be assign","authors_text":"Feng Xu, Jing-cheng Pang, Shengyi Jiang, Xu-Hui Liu, Yang Yu, Zhenghai Xue","cross_cats":["cs.AI"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-05-15T16:08:45Z","title":"Regret Minimization Experience Replay in Off-Policy Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2105.07253","kind":"arxiv","version":3},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:5dd10dc6ab006c879839287a03442d4cb9e49a6541f3777fcfc1c7508e28785a","target":"record","created_at":"2026-07-05T03:30:11Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"d04d4a0d2df4dc38858de2ff850b284786fbb154d48776022b3ca37e1bfdc000","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-05-15T16:08:45Z","title_canon_sha256":"57e7cf3ef10d2bf8517470e636bff10f829303bed7ab72226d031c8e86687c11"},"schema_version":"1.0","source":{"id":"2105.07253","kind":"arxiv","version":3}},"canonical_sha256":"cc063fdc53328e6a6cc8f9e7baffab4ad0119fafaa5f4f1f098a4c17ca14517b","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"cc063fdc53328e6a6cc8f9e7baffab4ad0119fafaa5f4f1f098a4c17ca14517b","first_computed_at":"2026-07-05T03:30:11.735566Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T03:30:11.735566Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"ld7Qk1nGmNqs9jUAZszlIuplsf6F9hHUB+uehOLr048K0vDDzzZu9+4vOzzoMjV6Q/Git1fXIETURrE5zZeLAw==","signature_status":"signed_v1","signed_at":"2026-07-05T03:30:11.735968Z","signed_message":"canonical_sha256_bytes"},"source_id":"2105.07253","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:5dd10dc6ab006c879839287a03442d4cb9e49a6541f3777fcfc1c7508e28785a","sha256:e0a0c9e8f61a5ff9853dbe803981c60009d2066de556bd40ab297c86e66567ea"],"state_sha256":"cb1a957f05bb4e178a464880820ada9965a7efaa1085b7fa2b8d844f6f8c078d"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"WTX+9a1pQ6GWlVbbG2sp+QoDUIKCPgZHEdOCyflrpyF2ga1fdUdUQiKWoZnSKbdC4MMYQZflxUvJkAvWCoHgDw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-07-24T19:25:13.951612Z","bundle_sha256":"16756c40a2c6a93ba61ff1accca733a3c1288e6c6a85025301014974070baee3"}}