{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2020:T65YNEXURDNFNSYUVFXHY3WZ6J","short_pith_number":"pith:T65YNEXU","canonical_record":{"source":{"id":"2005.01643","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-05-04T17:00:15Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"1c68a17cdece2a11fd06abef57849cb2a9451da6caef906abcd59e05a23fba2d","abstract_canon_sha256":"8d228669cca8ac6967647febbb392405270f57326c34c4ea9b46631c3d9f8a06"},"schema_version":"1.0"},"canonical_sha256":"9fbb8692f488da56cb14a96e7c6ed9f2400350479f649781eab504b1c7bd84ce","source":{"kind":"arxiv","id":"2005.01643","version":3},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2005.01643","created_at":"2026-07-05T01:48:09Z"},{"alias_kind":"arxiv_version","alias_value":"2005.01643v3","created_at":"2026-07-05T01:48:09Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2005.01643","created_at":"2026-07-05T01:48:09Z"},{"alias_kind":"pith_short_12","alias_value":"T65YNEXURDNF","created_at":"2026-07-05T01:48:09Z"},{"alias_kind":"pith_short_16","alias_value":"T65YNEXURDNFNSYU","created_at":"2026-07-05T01:48:09Z"},{"alias_kind":"pith_short_8","alias_value":"T65YNEXU","created_at":"2026-07-05T01:48:09Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2020:T65YNEXURDNFNSYUVFXHY3WZ6J","target":"record","payload":{"canonical_record":{"source":{"id":"2005.01643","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-05-04T17:00:15Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"1c68a17cdece2a11fd06abef57849cb2a9451da6caef906abcd59e05a23fba2d","abstract_canon_sha256":"8d228669cca8ac6967647febbb392405270f57326c34c4ea9b46631c3d9f8a06"},"schema_version":"1.0"},"canonical_sha256":"9fbb8692f488da56cb14a96e7c6ed9f2400350479f649781eab504b1c7bd84ce","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:48:09.320542Z","signature_b64":"LuUHyFFAU0D21VBQKxBUjOk2AJEIDi+hURhRh6D19i6r2H+CLII0dPIhWoGLZeD2yuUQTv2QwIdOEWJnaOR4Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9fbb8692f488da56cb14a96e7c6ed9f2400350479f649781eab504b1c7bd84ce","last_reissued_at":"2026-07-05T01:48:09.320037Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:48:09.320037Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2005.01643","source_version":3,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T01:48:09Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"9fBNeJoBeAQFHXuPW12l7ON40zwkp+Cq7K75y9MpZl0EDREoRGrzmcxrEA5doGQqJVVoU3/pTjF4N4WXg6kdDw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-05T14:58:45.036332Z"},"content_sha256":"31fc54aba7099886e1f0d9561f818e9531b67416a97c95f4fad6b78018b0a58e","schema_version":"1.0","event_id":"sha256:31fc54aba7099886e1f0d9561f818e9531b67416a97c95f4fad6b78018b0a58e"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2020:T65YNEXURDNFNSYUVFXHY3WZ6J","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Offline Reinforcement Learning: Tutorial, Review, and Perspectives on Open Problems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"Offline reinforcement learning can extract maximum-utility policies from fixed datasets without new data collection.","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Aviral Kumar, George Tucker, Justin Fu, Sergey Levine","submitted_at":"2020-05-04T17:00:15Z","abstract_excerpt":"In this tutorial article, we aim to provide the reader with the conceptual tools needed to get started on research on offline reinforcement learning algorithms: reinforcement learning algorithms that utilize previously collected data, without additional online data collection. Offline reinforcement learning algorithms hold tremendous promise for making it possible to turn large datasets into powerful decision making engines. Effective offline reinforcement learning methods would be able to extract policies with the maximum possible utility out of the available data, thereby allowing automation"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"Offline reinforcement learning algorithms hold tremendous promise for making it possible to turn large datasets into powerful decision making engines.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"That the limitations of current algorithms can be mitigated by solutions explored in recent work, allowing effective extraction of maximum-utility policies from available data.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"Offline RL promises to extract high-utility policies from static datasets but faces fundamental challenges that current methods only partially address.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"Offline reinforcement learning can extract maximum-utility policies from fixed datasets without new data collection.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"09e175c39437cf8093a48c27e47793cfbb82f1f4f3a2d8f442ed44412f218e30"},"source":{"id":"2005.01643","kind":"arxiv","version":3},"verdict":{"id":"6bfd068a-04da-4d25-a320-5da9f8adbee7","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-11T11:27:57.485356Z","strongest_claim":"Offline reinforcement learning algorithms hold tremendous promise for making it possible to turn large datasets into powerful decision making engines.","one_line_summary":"Offline RL promises to extract high-utility policies from static datasets but faces fundamental challenges that current methods only partially address.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"That the limitations of current algorithms can be mitigated by solutions explored in recent work, allowing effective extraction of maximum-utility policies from available data.","pith_extraction_headline":"Offline reinforcement learning can extract maximum-utility policies from fixed datasets without new data collection."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2005.01643/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":284,"sample":[{"doi":"","year":2009,"title":"Koller, D. and Friedman, N. , title =. 2009 , isbn =","work_id":"b4f13d1c-91fe-4f3c-9585-19f70b43e72b","ref_index":1,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2019,"title":"2019 International Conference on Robotics and Automation (ICRA) , pages=","work_id":"0e8149b3-cb9a-4833-83a0-14c5ca8a378d","ref_index":2,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2015,"title":"International journal of computer vision , volume=","work_id":"0d44db06-23e3-4a2d-b3e3-8c224934b42e","ref_index":3,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":null,"title":"Sim-to-Real: Learning Agile Locomotion For Quadruped Robots","work_id":"e0d353f4-f934-4d22-af2e-808140cf91c5","ref_index":4,"cited_arxiv_id":"1804.10332","is_internal_anchor":false},{"doi":"","year":null,"title":"Sadeghi, Fereshteh and Levine, Sergey , booktitle=","work_id":"4ff298b1-1d35-4fa6-a224-5cee8a077c65","ref_index":5,"cited_arxiv_id":"","is_internal_anchor":false}],"resolved_work":284,"snapshot_sha256":"c224ae8fa56c862c36c39d93fb7aca2d5358ee18e8a8e52507eccf615de9bbd5","internal_anchors":28},"formal_canon":{"evidence_count":3,"snapshot_sha256":"94f4dda78c90b203c240e3e3b54ddfdcf49bc85862b8cd4df90199116a0dbc02"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"6bfd068a-04da-4d25-a320-5da9f8adbee7"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T01:48:09Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"x74ZclRkcX5Zl4s02BG30ajcfGDQnkJP/CyUILIaXwS0P9urhcEWq7dQPLbOo+LliMtcXzPtj5Hh/Q+P/5rUCA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-05T14:58:45.037107Z"},"content_sha256":"8ff2f2d0eb4d5364820dda9e63901b2feaf372e4cd1d63f2c5f79b3b3f840ec6","schema_version":"1.0","event_id":"sha256:8ff2f2d0eb4d5364820dda9e63901b2feaf372e4cd1d63f2c5f79b3b3f840ec6"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/T65YNEXURDNFNSYUVFXHY3WZ6J/bundle.json","state_url":"https://pith.science/pith/T65YNEXURDNFNSYUVFXHY3WZ6J/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/T65YNEXURDNFNSYUVFXHY3WZ6J/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-05T14:58:45Z","links":{"resolver":"https://pith.science/pith/T65YNEXURDNFNSYUVFXHY3WZ6J","bundle":"https://pith.science/pith/T65YNEXURDNFNSYUVFXHY3WZ6J/bundle.json","state":"https://pith.science/pith/T65YNEXURDNFNSYUVFXHY3WZ6J/state.json","well_known_bundle":"https://pith.science/.well-known/pith/T65YNEXURDNFNSYUVFXHY3WZ6J/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2020:T65YNEXURDNFNSYUVFXHY3WZ6J","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"8d228669cca8ac6967647febbb392405270f57326c34c4ea9b46631c3d9f8a06","cross_cats_sorted":["cs.AI","stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-05-04T17:00:15Z","title_canon_sha256":"1c68a17cdece2a11fd06abef57849cb2a9451da6caef906abcd59e05a23fba2d"},"schema_version":"1.0","source":{"id":"2005.01643","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2005.01643","created_at":"2026-07-05T01:48:09Z"},{"alias_kind":"arxiv_version","alias_value":"2005.01643v3","created_at":"2026-07-05T01:48:09Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2005.01643","created_at":"2026-07-05T01:48:09Z"},{"alias_kind":"pith_short_12","alias_value":"T65YNEXURDNF","created_at":"2026-07-05T01:48:09Z"},{"alias_kind":"pith_short_16","alias_value":"T65YNEXURDNFNSYU","created_at":"2026-07-05T01:48:09Z"},{"alias_kind":"pith_short_8","alias_value":"T65YNEXU","created_at":"2026-07-05T01:48:09Z"}],"graph_snapshots":[{"event_id":"sha256:8ff2f2d0eb4d5364820dda9e63901b2feaf372e4cd1d63f2c5f79b3b3f840ec6","target":"graph","created_at":"2026-07-05T01:48:09Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"Offline reinforcement learning algorithms hold tremendous promise for making it possible to turn large datasets into powerful decision making engines."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"That the limitations of current algorithms can be mitigated by solutions explored in recent work, allowing effective extraction of maximum-utility policies from available data."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"Offline RL promises to extract high-utility policies from static datasets but faces fundamental challenges that current methods only partially address."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"Offline reinforcement learning can extract maximum-utility policies from fixed datasets without new data collection."}],"snapshot_sha256":"09e175c39437cf8093a48c27e47793cfbb82f1f4f3a2d8f442ed44412f218e30"},"formal_canon":{"evidence_count":3,"snapshot_sha256":"94f4dda78c90b203c240e3e3b54ddfdcf49bc85862b8cd4df90199116a0dbc02"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2005.01643/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"In this tutorial article, we aim to provide the reader with the conceptual tools needed to get started on research on offline reinforcement learning algorithms: reinforcement learning algorithms that utilize previously collected data, without additional online data collection. Offline reinforcement learning algorithms hold tremendous promise for making it possible to turn large datasets into powerful decision making engines. Effective offline reinforcement learning methods would be able to extract policies with the maximum possible utility out of the available data, thereby allowing automation","authors_text":"Aviral Kumar, George Tucker, Justin Fu, Sergey Levine","cross_cats":["cs.AI","stat.ML"],"headline":"Offline reinforcement learning can extract maximum-utility policies from fixed datasets without new data collection.","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-05-04T17:00:15Z","title":"Offline Reinforcement Learning: Tutorial, Review, and Perspectives on Open Problems"},"references":{"count":284,"internal_anchors":28,"resolved_work":284,"sample":[{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":1,"title":"Koller, D. and Friedman, N. , title =. 2009 , isbn =","work_id":"b4f13d1c-91fe-4f3c-9585-19f70b43e72b","year":2009},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":2,"title":"2019 International Conference on Robotics and Automation (ICRA) , pages=","work_id":"0e8149b3-cb9a-4833-83a0-14c5ca8a378d","year":2019},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":3,"title":"International journal of computer vision , volume=","work_id":"0d44db06-23e3-4a2d-b3e3-8c224934b42e","year":2015},{"cited_arxiv_id":"1804.10332","doi":"","is_internal_anchor":false,"ref_index":4,"title":"Sim-to-Real: Learning Agile Locomotion For Quadruped Robots","work_id":"e0d353f4-f934-4d22-af2e-808140cf91c5","year":null},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":5,"title":"Sadeghi, Fereshteh and Levine, Sergey , booktitle=","work_id":"4ff298b1-1d35-4fa6-a224-5cee8a077c65","year":null}],"snapshot_sha256":"c224ae8fa56c862c36c39d93fb7aca2d5358ee18e8a8e52507eccf615de9bbd5"},"source":{"id":"2005.01643","kind":"arxiv","version":3},"verdict":{"created_at":"2026-05-11T11:27:57.485356Z","id":"6bfd068a-04da-4d25-a320-5da9f8adbee7","model_set":{"reader":"grok-4.3"},"one_line_summary":"Offline RL promises to extract high-utility policies from static datasets but faces fundamental challenges that current methods only partially address.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"Offline reinforcement learning can extract maximum-utility policies from fixed datasets without new data collection.","strongest_claim":"Offline reinforcement learning algorithms hold tremendous promise for making it possible to turn large datasets into powerful decision making engines.","weakest_assumption":"That the limitations of current algorithms can be mitigated by solutions explored in recent work, allowing effective extraction of maximum-utility policies from available data."}},"verdict_id":"6bfd068a-04da-4d25-a320-5da9f8adbee7"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:31fc54aba7099886e1f0d9561f818e9531b67416a97c95f4fad6b78018b0a58e","target":"record","created_at":"2026-07-05T01:48:09Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"8d228669cca8ac6967647febbb392405270f57326c34c4ea9b46631c3d9f8a06","cross_cats_sorted":["cs.AI","stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-05-04T17:00:15Z","title_canon_sha256":"1c68a17cdece2a11fd06abef57849cb2a9451da6caef906abcd59e05a23fba2d"},"schema_version":"1.0","source":{"id":"2005.01643","kind":"arxiv","version":3}},"canonical_sha256":"9fbb8692f488da56cb14a96e7c6ed9f2400350479f649781eab504b1c7bd84ce","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"9fbb8692f488da56cb14a96e7c6ed9f2400350479f649781eab504b1c7bd84ce","first_computed_at":"2026-07-05T01:48:09.320037Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T01:48:09.320037Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"LuUHyFFAU0D21VBQKxBUjOk2AJEIDi+hURhRh6D19i6r2H+CLII0dPIhWoGLZeD2yuUQTv2QwIdOEWJnaOR4Cg==","signature_status":"signed_v1","signed_at":"2026-07-05T01:48:09.320542Z","signed_message":"canonical_sha256_bytes"},"source_id":"2005.01643","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:31fc54aba7099886e1f0d9561f818e9531b67416a97c95f4fad6b78018b0a58e","sha256:8ff2f2d0eb4d5364820dda9e63901b2feaf372e4cd1d63f2c5f79b3b3f840ec6"],"state_sha256":"5dd633278b7b1d55e4537b1bee70c648ea62c7ff4df1760c7a6816cfff30413b"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"0d05YgI/I8+Iefjzu2ohvM4Bzxjf8mtoV8lqIxjoj7H/81dBmMm0xKIIyY7NOL5MT7JW2grDAANN904Ay9afCw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-05T14:58:45.041590Z","bundle_sha256":"53e70f9cebd9b63aac9036596b3824c96673dee81ca0eb3c1bcca3b38480bac1"}}