{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:A44HYHPGPONVHOT3K26PEH7JL2","short_pith_number":"pith:A44HYHPG","canonical_record":{"source":{"id":"2506.09270","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-10T22:07:13Z","cross_cats_sorted":[],"title_canon_sha256":"3e2f7be056cd277e0dac8cddfd56924877b70541b9f7d81ceeef892372a5f117","abstract_canon_sha256":"420a1b226afac1cbb62b18803d336a788b0e4aaae6922f3e4c20fc3fcaf586e7"},"schema_version":"1.0"},"canonical_sha256":"07387c1de67b9b53ba7b56bcf21fe95e8fc9263cb9a8361b17a6e34d3e1368f2","source":{"kind":"arxiv","id":"2506.09270","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2506.09270","created_at":"2026-07-05T11:19:33Z"},{"alias_kind":"arxiv_version","alias_value":"2506.09270v1","created_at":"2026-07-05T11:19:33Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.09270","created_at":"2026-07-05T11:19:33Z"},{"alias_kind":"pith_short_12","alias_value":"A44HYHPGPONV","created_at":"2026-07-05T11:19:33Z"},{"alias_kind":"pith_short_16","alias_value":"A44HYHPGPONVHOT3","created_at":"2026-07-05T11:19:33Z"},{"alias_kind":"pith_short_8","alias_value":"A44HYHPG","created_at":"2026-07-05T11:19:33Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:A44HYHPGPONVHOT3K26PEH7JL2","target":"record","payload":{"canonical_record":{"source":{"id":"2506.09270","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-10T22:07:13Z","cross_cats_sorted":[],"title_canon_sha256":"3e2f7be056cd277e0dac8cddfd56924877b70541b9f7d81ceeef892372a5f117","abstract_canon_sha256":"420a1b226afac1cbb62b18803d336a788b0e4aaae6922f3e4c20fc3fcaf586e7"},"schema_version":"1.0"},"canonical_sha256":"07387c1de67b9b53ba7b56bcf21fe95e8fc9263cb9a8361b17a6e34d3e1368f2","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:19:33.956225Z","signature_b64":"+o+WwNksz4D5hXerTNpRcAhCvuVjNPyqVPQR6Q6aXhUB5M3HWVD+Rxfwko2joCDiL482coBW97xJs0fEn15uBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"07387c1de67b9b53ba7b56bcf21fe95e8fc9263cb9a8361b17a6e34d3e1368f2","last_reissued_at":"2026-07-05T11:19:33.955702Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:19:33.955702Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2506.09270","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:19:33Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"VyBamkc7LyENXADIR4o/yYBgbqSu8EwlPIvKWbU1NkUUXJ9HbAhR1y0NIZ5rPG9hV0ndGRvWR3pjgKZf+VroCg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-18T18:19:16.945897Z"},"content_sha256":"70af890f172b50638e8f9069fcd3553d7df0ba61a688d0bde0acb139314e1b82","schema_version":"1.0","event_id":"sha256:70af890f172b50638e8f9069fcd3553d7df0ba61a688d0bde0acb139314e1b82"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:A44HYHPGPONVHOT3K26PEH7JL2","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Uncertainty Prioritized Experience Replay","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Claudia Clopath, Rodrigo Carrasco-Davis, Sebastian Lee, Will Dabney","submitted_at":"2025-06-10T22:07:13Z","abstract_excerpt":"Prioritized experience replay, which improves sample efficiency by selecting relevant transitions to update parameter estimates, is a crucial component of contemporary value-based deep reinforcement learning models. Typically, transitions are prioritized based on their temporal difference error. However, this approach is prone to favoring noisy transitions, even when the value estimation closely approximates the target mean. This phenomenon resembles the noisy TV problem postulated in the exploration literature, in which exploration-guided agents get stuck by mistaking noise for novelty. To mi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.09270","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.09270/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:19:33Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"5NaqAieUgc3KNBts41BqzQ8SxR5g3Bzy4Twj3B49NRYsmuOXynZmn43z4cmtr2p8JwuvquhfHBPBwH6d3vbdBg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-18T18:19:16.946482Z"},"content_sha256":"5ba639372cd2ce590126ee4d52227545e209724f73b33e8ab7df031ebeb96e7b","schema_version":"1.0","event_id":"sha256:5ba639372cd2ce590126ee4d52227545e209724f73b33e8ab7df031ebeb96e7b"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/A44HYHPGPONVHOT3K26PEH7JL2/bundle.json","state_url":"https://pith.science/pith/A44HYHPGPONVHOT3K26PEH7JL2/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/A44HYHPGPONVHOT3K26PEH7JL2/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-18T18:19:16Z","links":{"resolver":"https://pith.science/pith/A44HYHPGPONVHOT3K26PEH7JL2","bundle":"https://pith.science/pith/A44HYHPGPONVHOT3K26PEH7JL2/bundle.json","state":"https://pith.science/pith/A44HYHPGPONVHOT3K26PEH7JL2/state.json","well_known_bundle":"https://pith.science/.well-known/pith/A44HYHPGPONVHOT3K26PEH7JL2/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:A44HYHPGPONVHOT3K26PEH7JL2","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"420a1b226afac1cbb62b18803d336a788b0e4aaae6922f3e4c20fc3fcaf586e7","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-10T22:07:13Z","title_canon_sha256":"3e2f7be056cd277e0dac8cddfd56924877b70541b9f7d81ceeef892372a5f117"},"schema_version":"1.0","source":{"id":"2506.09270","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2506.09270","created_at":"2026-07-05T11:19:33Z"},{"alias_kind":"arxiv_version","alias_value":"2506.09270v1","created_at":"2026-07-05T11:19:33Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.09270","created_at":"2026-07-05T11:19:33Z"},{"alias_kind":"pith_short_12","alias_value":"A44HYHPGPONV","created_at":"2026-07-05T11:19:33Z"},{"alias_kind":"pith_short_16","alias_value":"A44HYHPGPONVHOT3","created_at":"2026-07-05T11:19:33Z"},{"alias_kind":"pith_short_8","alias_value":"A44HYHPG","created_at":"2026-07-05T11:19:33Z"}],"graph_snapshots":[{"event_id":"sha256:5ba639372cd2ce590126ee4d52227545e209724f73b33e8ab7df031ebeb96e7b","target":"graph","created_at":"2026-07-05T11:19:33Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2506.09270/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Prioritized experience replay, which improves sample efficiency by selecting relevant transitions to update parameter estimates, is a crucial component of contemporary value-based deep reinforcement learning models. Typically, transitions are prioritized based on their temporal difference error. However, this approach is prone to favoring noisy transitions, even when the value estimation closely approximates the target mean. This phenomenon resembles the noisy TV problem postulated in the exploration literature, in which exploration-guided agents get stuck by mistaking noise for novelty. To mi","authors_text":"Claudia Clopath, Rodrigo Carrasco-Davis, Sebastian Lee, Will Dabney","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-10T22:07:13Z","title":"Uncertainty Prioritized Experience Replay"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.09270","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:70af890f172b50638e8f9069fcd3553d7df0ba61a688d0bde0acb139314e1b82","target":"record","created_at":"2026-07-05T11:19:33Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"420a1b226afac1cbb62b18803d336a788b0e4aaae6922f3e4c20fc3fcaf586e7","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-10T22:07:13Z","title_canon_sha256":"3e2f7be056cd277e0dac8cddfd56924877b70541b9f7d81ceeef892372a5f117"},"schema_version":"1.0","source":{"id":"2506.09270","kind":"arxiv","version":1}},"canonical_sha256":"07387c1de67b9b53ba7b56bcf21fe95e8fc9263cb9a8361b17a6e34d3e1368f2","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"07387c1de67b9b53ba7b56bcf21fe95e8fc9263cb9a8361b17a6e34d3e1368f2","first_computed_at":"2026-07-05T11:19:33.955702Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:19:33.955702Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"+o+WwNksz4D5hXerTNpRcAhCvuVjNPyqVPQR6Q6aXhUB5M3HWVD+Rxfwko2joCDiL482coBW97xJs0fEn15uBQ==","signature_status":"signed_v1","signed_at":"2026-07-05T11:19:33.956225Z","signed_message":"canonical_sha256_bytes"},"source_id":"2506.09270","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:70af890f172b50638e8f9069fcd3553d7df0ba61a688d0bde0acb139314e1b82","sha256:5ba639372cd2ce590126ee4d52227545e209724f73b33e8ab7df031ebeb96e7b"],"state_sha256":"d61f63b174d57a1d046d8fc7585bd56f07c56309d5d986a0ff3315ffa9bcf2b3"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Na2hkoHEXBp/78MUqun8SMY8iUxeT+U/Bq1ImnLIMlL5ONdMjI1XcS+TVMtMzT5HDU3kHk7PcENuxwwBLQKXDA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-18T18:19:16.951814Z","bundle_sha256":"59a93864441f99c09ce36918e3806b37fb800ea9e21063940afaf82e5c583e53"}}