{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:QPFJAI6BC3GK7WJR4OOVYP4BXI","short_pith_number":"pith:QPFJAI6B","schema_version":"1.0","canonical_sha256":"83ca9023c116ccafd931e39d5c3f81ba0f005fc4f1ea106d844e8a2c1ac03459","source":{"kind":"arxiv","id":"1905.12726","version":2},"attestation_state":"computed","paper":{"title":"Prioritized Sequence Experience Replay","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Josh Bertram, Marc Brittain, Peng Wei, Xuxi Yang","submitted_at":"2019-05-25T15:38:00Z","abstract_excerpt":"Experience replay is widely used in deep reinforcement learning algorithms and allows agents to remember and learn from experiences from the past. In an effort to learn more efficiently, researchers proposed prioritized experience replay (PER) which samples important transitions more frequently. In this paper, we propose Prioritized Sequence Experience Replay (PSER) a framework for prioritizing sequences of experience in an attempt to both learn more efficiently and to obtain better performance. We compare the performance of PER and PSER sampling techniques in a tabular Q-learning environment "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1905.12726","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-05-25T15:38:00Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"b88cac9b6428efd1bb99431692eed720837eed2f1afb295d59b16642535879cf","abstract_canon_sha256":"47bc9552750e9f4e41b4dbb3bac57b41bf4a693950276679c9e3405f26a4f6b2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:42:19.325943Z","signature_b64":"zBcmXNoz02Dyot3lQqnDMXxEkjJnGEkB8uZXw99ujSD3aLnAsQzrPa6hARHMCKbluz9POV6M+3t7lRu+9Nz4Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"83ca9023c116ccafd931e39d5c3f81ba0f005fc4f1ea106d844e8a2c1ac03459","last_reissued_at":"2026-07-05T00:42:19.325549Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:42:19.325549Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Prioritized Sequence Experience Replay","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Josh Bertram, Marc Brittain, Peng Wei, Xuxi Yang","submitted_at":"2019-05-25T15:38:00Z","abstract_excerpt":"Experience replay is widely used in deep reinforcement learning algorithms and allows agents to remember and learn from experiences from the past. In an effort to learn more efficiently, researchers proposed prioritized experience replay (PER) which samples important transitions more frequently. In this paper, we propose Prioritized Sequence Experience Replay (PSER) a framework for prioritizing sequences of experience in an attempt to both learn more efficiently and to obtain better performance. We compare the performance of PER and PSER sampling techniques in a tabular Q-learning environment "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1905.12726","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1905.12726/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1905.12726","created_at":"2026-07-05T00:42:19.325607+00:00"},{"alias_kind":"arxiv_version","alias_value":"1905.12726v2","created_at":"2026-07-05T00:42:19.325607+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1905.12726","created_at":"2026-07-05T00:42:19.325607+00:00"},{"alias_kind":"pith_short_12","alias_value":"QPFJAI6BC3GK","created_at":"2026-07-05T00:42:19.325607+00:00"},{"alias_kind":"pith_short_16","alias_value":"QPFJAI6BC3GK7WJR","created_at":"2026-07-05T00:42:19.325607+00:00"},{"alias_kind":"pith_short_8","alias_value":"QPFJAI6B","created_at":"2026-07-05T00:42:19.325607+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2501.18093","citing_title":"Reward Prediction Error Prioritisation in Experience Replay: The RPE-PER Method","ref_index":2,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QPFJAI6BC3GK7WJR4OOVYP4BXI","json":"https://pith.science/pith/QPFJAI6BC3GK7WJR4OOVYP4BXI.json","graph_json":"https://pith.science/api/pith-number/QPFJAI6BC3GK7WJR4OOVYP4BXI/graph.json","events_json":"https://pith.science/api/pith-number/QPFJAI6BC3GK7WJR4OOVYP4BXI/events.json","paper":"https://pith.science/paper/QPFJAI6B"},"agent_actions":{"view_html":"https://pith.science/pith/QPFJAI6BC3GK7WJR4OOVYP4BXI","download_json":"https://pith.science/pith/QPFJAI6BC3GK7WJR4OOVYP4BXI.json","view_paper":"https://pith.science/paper/QPFJAI6B","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1905.12726&json=true","fetch_graph":"https://pith.science/api/pith-number/QPFJAI6BC3GK7WJR4OOVYP4BXI/graph.json","fetch_events":"https://pith.science/api/pith-number/QPFJAI6BC3GK7WJR4OOVYP4BXI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QPFJAI6BC3GK7WJR4OOVYP4BXI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QPFJAI6BC3GK7WJR4OOVYP4BXI/action/storage_attestation","attest_author":"https://pith.science/pith/QPFJAI6BC3GK7WJR4OOVYP4BXI/action/author_attestation","sign_citation":"https://pith.science/pith/QPFJAI6BC3GK7WJR4OOVYP4BXI/action/citation_signature","submit_replication":"https://pith.science/pith/QPFJAI6BC3GK7WJR4OOVYP4BXI/action/replication_record"}},"created_at":"2026-07-05T00:42:19.325607+00:00","updated_at":"2026-07-05T00:42:19.325607+00:00"}