{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2022:JVDAIWAWOLPUPXIXR5EJ62LBI4","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"e19ca0de18f33c10f01e62d0ec6f137f647dc1f2cfd9837a51da0a8f34d03d81","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-02-23T15:27:16Z","title_canon_sha256":"c8f3e10caf03092d3bac029fe7d87051d024acf1bd750980d3b94b3409e848a3"},"schema_version":"1.0","source":{"id":"2202.11566","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2202.11566","created_at":"2026-07-05T03:59:34Z"},{"alias_kind":"arxiv_version","alias_value":"2202.11566v1","created_at":"2026-07-05T03:59:34Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2202.11566","created_at":"2026-07-05T03:59:34Z"},{"alias_kind":"pith_short_12","alias_value":"JVDAIWAWOLPU","created_at":"2026-07-05T03:59:34Z"},{"alias_kind":"pith_short_16","alias_value":"JVDAIWAWOLPUPXIX","created_at":"2026-07-05T03:59:34Z"},{"alias_kind":"pith_short_8","alias_value":"JVDAIWAW","created_at":"2026-07-05T03:59:34Z"}],"graph_snapshots":[{"event_id":"sha256:b0e86723168eabdbb061049c7db91b2a5e3556d741a5b8c18931bd514c3d86cd","target":"graph","created_at":"2026-07-05T03:59:34Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2202.11566/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Offline Reinforcement Learning (RL) aims to learn policies from previously collected datasets without exploring the environment. Directly applying off-policy algorithms to offline RL usually fails due to the extrapolation error caused by the out-of-distribution (OOD) actions. Previous methods tackle such problem by penalizing the Q-values of OOD actions or constraining the trained policy to be close to the behavior policy. Nevertheless, such methods typically prevent the generalization of value functions beyond the offline data and also lack precise characterization of OOD data. In this paper,","authors_text":"Animesh Garg, Chenjia Bai, Lingxiao Wang, Peng Liu, Zhaoran Wang, Zhihong Deng, Zhuoran Yang","cross_cats":[],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-02-23T15:27:16Z","title":"Pessimistic Bootstrapping for Uncertainty-Driven Offline Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2202.11566","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:72b41c7c90d32897494355fe5cc2f0e37b415b0b2c2d5783c7e098f0748f7ef9","target":"record","created_at":"2026-07-05T03:59:34Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"e19ca0de18f33c10f01e62d0ec6f137f647dc1f2cfd9837a51da0a8f34d03d81","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-02-23T15:27:16Z","title_canon_sha256":"c8f3e10caf03092d3bac029fe7d87051d024acf1bd750980d3b94b3409e848a3"},"schema_version":"1.0","source":{"id":"2202.11566","kind":"arxiv","version":1}},"canonical_sha256":"4d4604581672df47dd178f489f69614722e2e1361ffa7bd1ef4c1c733c0e86af","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"4d4604581672df47dd178f489f69614722e2e1361ffa7bd1ef4c1c733c0e86af","first_computed_at":"2026-07-05T03:59:34.499223Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T03:59:34.499223Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"V9HwLB29MyEsKgms/QrVDB+JCNGBAzkXQpBeeRGRd4BTZEoqq39jLHT1Y/nOX+MEaJjRAdfsBgcMTdkb0JIvAg==","signature_status":"signed_v1","signed_at":"2026-07-05T03:59:34.499707Z","signed_message":"canonical_sha256_bytes"},"source_id":"2202.11566","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:72b41c7c90d32897494355fe5cc2f0e37b415b0b2c2d5783c7e098f0748f7ef9","sha256:b0e86723168eabdbb061049c7db91b2a5e3556d741a5b8c18931bd514c3d86cd"],"state_sha256":"4d89cdbafe8c3fe99114ff0470bbfcf8c5bcfc281e3b8db24b815679ee975e97"}