{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2020:D2SNJQKLI5QUGBTXUQYQSF73PR","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"5512cf4724e468ca0a0f4414d6d074c3f6cdb4aae9db977fb8f204e54855c16b","cross_cats_sorted":["stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-04-15T17:18:19Z","title_canon_sha256":"35a4c1a61624d1cfe16366f887d9938531d3185d83679fea4470f32dcc26b27f"},"schema_version":"1.0","source":{"id":"2004.07219","kind":"arxiv","version":4}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2004.07219","created_at":"2026-07-05T02:13:13Z"},{"alias_kind":"arxiv_version","alias_value":"2004.07219v4","created_at":"2026-07-05T02:13:13Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2004.07219","created_at":"2026-07-05T02:13:13Z"},{"alias_kind":"pith_short_12","alias_value":"D2SNJQKLI5QU","created_at":"2026-07-05T02:13:13Z"},{"alias_kind":"pith_short_16","alias_value":"D2SNJQKLI5QUGBTX","created_at":"2026-07-05T02:13:13Z"},{"alias_kind":"pith_short_8","alias_value":"D2SNJQKL","created_at":"2026-07-05T02:13:13Z"}],"graph_snapshots":[{"event_id":"sha256:b453f534669f2f9da4790196c5dea2f2d84170f13da2284afdbe4f145a390cbb","target":"graph","created_at":"2026-07-05T02:13:13Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"By moving beyond simple benchmark tasks and data collected by partially-trained RL agents, we reveal important and unappreciated deficiencies of existing algorithms."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"That datasets generated via hand-designed controllers, human demonstrators, multitask settings, and mixtures of policies capture the key properties most relevant to real-world offline RL applications."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"D4RL supplies new offline RL benchmarks and datasets from expert and mixed sources to expose weaknesses in existing algorithms and standardize evaluation."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"New benchmark datasets for offline RL, drawn from human demonstrations and mixed policies, expose deficiencies in existing algorithms."}],"snapshot_sha256":"0c06905e9ba24f8729ba91ab0b61bdb9ea59af869a8a84e57b82b8486d7ef737"},"formal_canon":{"evidence_count":1,"snapshot_sha256":"896b842cd86dfef1170675a52ee16c796e1c4a6ec93fe924fa49b97263993227"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2004.07219/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"The offline reinforcement learning (RL) setting (also known as full batch RL), where a policy is learned from a static dataset, is compelling as progress enables RL methods to take advantage of large, previously-collected datasets, much like how the rise of large datasets has fueled results in supervised learning. However, existing online RL benchmarks are not tailored towards the offline setting and existing offline RL benchmarks are restricted to data generated by partially-trained agents, making progress in offline RL difficult to measure. In this work, we introduce benchmarks specifically ","authors_text":"Aviral Kumar, George Tucker, Justin Fu, Ofir Nachum, Sergey Levine","cross_cats":["stat.ML"],"headline":"New benchmark datasets for offline RL, drawn from human demonstrations and mixed policies, expose deficiencies in existing algorithms.","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-04-15T17:18:19Z","title":"D4RL: Datasets for Deep Data-Driven Reinforcement Learning"},"references":{"count":24,"internal_anchors":7,"resolved_work":24,"sample":[{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":1,"title":"Preprint arXiv:1908.00261 , year=","work_id":"f49e0fd8-840b-4c3d-bf03-e057e1da8f2d","year":1908},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":2,"title":"Scaling data-driven robotics with reward sketching and batch reinforcement learning.Preprint arXiv:1909.12200","work_id":"c44fff83-b8c4-4bf1-a47d-0f5c0046b36c","year":1909},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":3,"title":"End- to-end driving via conditional imitation learning","work_id":"a28935f5-b348-47b2-9aeb-25c303b9a4aa","year":2018},{"cited_arxiv_id":"1904.12901","doi":"","is_internal_anchor":false,"ref_index":4,"title":"Challenges of Real-World Reinforcement Learning","work_id":"fc99449a-80f4-4f37-a028-7b3774c78bf6","year":1904},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":5,"title":"Mankowitz, Jerry Li, Cosmin Paduraru, Sven Gowal, and Todd Hes- ter","work_id":"72539193-de5d-4093-a85f-abb4eba8699e","year":2003}],"snapshot_sha256":"680ebf13a40fb439d9b1b23727eb2e41ac080d99d079f74ccf5b59c7c056f646"},"source":{"id":"2004.07219","kind":"arxiv","version":4},"verdict":{"created_at":"2026-05-12T23:14:59.570931Z","id":"f9c60159-366b-4c1b-bde8-534bda9e44c0","model_set":{"reader":"grok-4.3"},"one_line_summary":"D4RL supplies new offline RL benchmarks and datasets from expert and mixed sources to expose weaknesses in existing algorithms and standardize evaluation.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"New benchmark datasets for offline RL, drawn from human demonstrations and mixed policies, expose deficiencies in existing algorithms.","strongest_claim":"By moving beyond simple benchmark tasks and data collected by partially-trained RL agents, we reveal important and unappreciated deficiencies of existing algorithms.","weakest_assumption":"That datasets generated via hand-designed controllers, human demonstrators, multitask settings, and mixtures of policies capture the key properties most relevant to real-world offline RL applications."}},"verdict_id":"f9c60159-366b-4c1b-bde8-534bda9e44c0"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:4143bbaa6d68067c4ad574a3a16f521aeef935e7ee9baca40a15ef289afb5f82","target":"record","created_at":"2026-07-05T02:13:13Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"5512cf4724e468ca0a0f4414d6d074c3f6cdb4aae9db977fb8f204e54855c16b","cross_cats_sorted":["stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-04-15T17:18:19Z","title_canon_sha256":"35a4c1a61624d1cfe16366f887d9938531d3185d83679fea4470f32dcc26b27f"},"schema_version":"1.0","source":{"id":"2004.07219","kind":"arxiv","version":4}},"canonical_sha256":"1ea4d4c14b4761430677a4310917fb7c736cb2b5838d04c44cb6c94c99428dbc","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"1ea4d4c14b4761430677a4310917fb7c736cb2b5838d04c44cb6c94c99428dbc","first_computed_at":"2026-07-05T02:13:13.910791Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T02:13:13.910791Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"GLSdUgMaJt28mxDV0vbKIpVJJ9UeXywLOq+aeVwRuG3BMDx97wIhY6ZftnLv8FbulBfLmiNj/BT1cLIpZdEICA==","signature_status":"signed_v1","signed_at":"2026-07-05T02:13:13.911233Z","signed_message":"canonical_sha256_bytes"},"source_id":"2004.07219","source_kind":"arxiv","source_version":4}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:4143bbaa6d68067c4ad574a3a16f521aeef935e7ee9baca40a15ef289afb5f82","sha256:b453f534669f2f9da4790196c5dea2f2d84170f13da2284afdbe4f145a390cbb"],"state_sha256":"daf15bcf7dc321f2af1012f1eacf8b57991620a8d8ba5a3a152e6be8fe4bd8f4"}