{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2021:KPUQRCQTBEIGYIK3LPA33JTTYT","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"b515fd128856e0d2421938b55517d43c92cc80ed24cb1afccef65caa3d9d6e93","cross_cats_sorted":["stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-02-18T08:54:14Z","title_canon_sha256":"a8c230c765a020484977719d5fd09fc87a46fc8df5b09cc8df2fd65fbd3f6e2d"},"schema_version":"1.0","source":{"id":"2102.09225","kind":"arxiv","version":4}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2102.09225","created_at":"2026-07-05T03:37:40Z"},{"alias_kind":"arxiv_version","alias_value":"2102.09225v4","created_at":"2026-07-05T03:37:40Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2102.09225","created_at":"2026-07-05T03:37:40Z"},{"alias_kind":"pith_short_12","alias_value":"KPUQRCQTBEIG","created_at":"2026-07-05T03:37:40Z"},{"alias_kind":"pith_short_16","alias_value":"KPUQRCQTBEIGYIK3","created_at":"2026-07-05T03:37:40Z"},{"alias_kind":"pith_short_8","alias_value":"KPUQRCQT","created_at":"2026-07-05T03:37:40Z"}],"graph_snapshots":[{"event_id":"sha256:9186094288f419fca4d7bcc223eb1afbfe32cf193071cc0361f4dd9e737ec654","target":"graph","created_at":"2026-07-05T03:37:40Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2102.09225/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reliant on too many experiments to learn good actions, current Reinforcement Learning (RL) algorithms have limited applicability in real-world settings, which can be too expensive to allow exploration. We propose an algorithm for batch RL, where effective policies are learned using only a fixed offline dataset instead of online interactions with the environment. The limited data in batch RL produces inherent uncertainty in value estimates of states/actions that were insufficiently represented in the training data. This leads to particularly severe extrapolation when our candidate policies dive","authors_text":"Alexander J. Smola, Jonas Mueller, Kavosh Asadi, Pratik Chaudhari, Rasool Fakoor","cross_cats":["stat.ML"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-02-18T08:54:14Z","title":"Continuous Doubly Constrained Batch Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2102.09225","kind":"arxiv","version":4},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:1790708a6caaa9a6d4cb50e774e7e263822e72ca004566cad9d1ad75d774b54b","target":"record","created_at":"2026-07-05T03:37:40Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"b515fd128856e0d2421938b55517d43c92cc80ed24cb1afccef65caa3d9d6e93","cross_cats_sorted":["stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-02-18T08:54:14Z","title_canon_sha256":"a8c230c765a020484977719d5fd09fc87a46fc8df5b09cc8df2fd65fbd3f6e2d"},"schema_version":"1.0","source":{"id":"2102.09225","kind":"arxiv","version":4}},"canonical_sha256":"53e9088a1309106c215b5bc1bda673c4d0a3150d9fb174b7813b43ad687f1853","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"53e9088a1309106c215b5bc1bda673c4d0a3150d9fb174b7813b43ad687f1853","first_computed_at":"2026-07-05T03:37:40.771667Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T03:37:40.771667Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"dT/VKEO/mmNVAWzPm4fWBL/mzHR3otESQxRV6tpsH1fi/oOSELT6EPQKW+BvBQjSsS32CtMQlBa+LuoIIjdrBQ==","signature_status":"signed_v1","signed_at":"2026-07-05T03:37:40.772184Z","signed_message":"canonical_sha256_bytes"},"source_id":"2102.09225","source_kind":"arxiv","source_version":4}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:1790708a6caaa9a6d4cb50e774e7e263822e72ca004566cad9d1ad75d774b54b","sha256:9186094288f419fca4d7bcc223eb1afbfe32cf193071cc0361f4dd9e737ec654"],"state_sha256":"ccd248573233f2e5491ef960968db48afc57a3130b5d2bef88a2904be44c4d8a"}