{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:TI3NK364C3PETPRVUR3Q5INJ22","short_pith_number":"pith:TI3NK364","schema_version":"1.0","canonical_sha256":"9a36d56fdc16de49be35a4770ea1a9d680ceefc8c17ade617bd2c618132f8273","source":{"kind":"arxiv","id":"2303.06614","version":4},"attestation_state":"computed","paper":{"title":"Synthetic Experience Replay","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Cong Lu, Jack Parker-Holder, Philip J. Ball, Yee Whye Teh","submitted_at":"2023-03-12T09:10:45Z","abstract_excerpt":"A key theme in the past decade has been that when large neural networks and large datasets combine they can produce remarkable results. In deep reinforcement learning (RL), this paradigm is commonly made possible through experience replay, whereby a dataset of past experiences is used to train a policy or value function. However, unlike in supervised or self-supervised learning, an RL agent has to collect its own data, which is often limited. Thus, it is challenging to reap the benefits of deep learning, and even small neural networks can overfit at the start of training. In this work, we leve"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.06614","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-03-12T09:10:45Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"421d57ff832057f231da131a929b492ce8e152fac6164a02d76d360ef0e5864e","abstract_canon_sha256":"7cb64fcf33e463613200846b714627d5144420bc6dea0a6f95f38136a73f5c43"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:05:41.744505Z","signature_b64":"7zE4fJpGr7PQKiSehvx1Dz7hbC+PZON+M1tIVCK70P61WnNeWgvlAvpsTLhzMe9NTZPChIqcjPIYxAbSe/4WCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9a36d56fdc16de49be35a4770ea1a9d680ceefc8c17ade617bd2c618132f8273","last_reissued_at":"2026-07-05T07:05:41.743998Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:05:41.743998Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Synthetic Experience Replay","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Cong Lu, Jack Parker-Holder, Philip J. Ball, Yee Whye Teh","submitted_at":"2023-03-12T09:10:45Z","abstract_excerpt":"A key theme in the past decade has been that when large neural networks and large datasets combine they can produce remarkable results. In deep reinforcement learning (RL), this paradigm is commonly made possible through experience replay, whereby a dataset of past experiences is used to train a policy or value function. However, unlike in supervised or self-supervised learning, an RL agent has to collect its own data, which is often limited. Thus, it is challenging to reap the benefits of deep learning, and even small neural networks can overfit at the start of training. In this work, we leve"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.06614","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.06614/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.06614","created_at":"2026-07-05T07:05:41.744059+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.06614v4","created_at":"2026-07-05T07:05:41.744059+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.06614","created_at":"2026-07-05T07:05:41.744059+00:00"},{"alias_kind":"pith_short_12","alias_value":"TI3NK364C3PE","created_at":"2026-07-05T07:05:41.744059+00:00"},{"alias_kind":"pith_short_16","alias_value":"TI3NK364C3PETPRV","created_at":"2026-07-05T07:05:41.744059+00:00"},{"alias_kind":"pith_short_8","alias_value":"TI3NK364","created_at":"2026-07-05T07:05:41.744059+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07693","citing_title":"Selective Timestep Weighting and Advantage-Based Replay for Sample-Efficient Diffusion RLHF","ref_index":22,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TI3NK364C3PETPRVUR3Q5INJ22","json":"https://pith.science/pith/TI3NK364C3PETPRVUR3Q5INJ22.json","graph_json":"https://pith.science/api/pith-number/TI3NK364C3PETPRVUR3Q5INJ22/graph.json","events_json":"https://pith.science/api/pith-number/TI3NK364C3PETPRVUR3Q5INJ22/events.json","paper":"https://pith.science/paper/TI3NK364"},"agent_actions":{"view_html":"https://pith.science/pith/TI3NK364C3PETPRVUR3Q5INJ22","download_json":"https://pith.science/pith/TI3NK364C3PETPRVUR3Q5INJ22.json","view_paper":"https://pith.science/paper/TI3NK364","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.06614&json=true","fetch_graph":"https://pith.science/api/pith-number/TI3NK364C3PETPRVUR3Q5INJ22/graph.json","fetch_events":"https://pith.science/api/pith-number/TI3NK364C3PETPRVUR3Q5INJ22/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TI3NK364C3PETPRVUR3Q5INJ22/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TI3NK364C3PETPRVUR3Q5INJ22/action/storage_attestation","attest_author":"https://pith.science/pith/TI3NK364C3PETPRVUR3Q5INJ22/action/author_attestation","sign_citation":"https://pith.science/pith/TI3NK364C3PETPRVUR3Q5INJ22/action/citation_signature","submit_replication":"https://pith.science/pith/TI3NK364C3PETPRVUR3Q5INJ22/action/replication_record"}},"created_at":"2026-07-05T07:05:41.744059+00:00","updated_at":"2026-07-05T07:05:41.744059+00:00"}