{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:JVDAIWAWOLPUPXIXR5EJ62LBI4","short_pith_number":"pith:JVDAIWAW","schema_version":"1.0","canonical_sha256":"4d4604581672df47dd178f489f69614722e2e1361ffa7bd1ef4c1c733c0e86af","source":{"kind":"arxiv","id":"2202.11566","version":1},"attestation_state":"computed","paper":{"title":"Pessimistic Bootstrapping for Uncertainty-Driven Offline Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Animesh Garg, Chenjia Bai, Lingxiao Wang, Peng Liu, Zhaoran Wang, Zhihong Deng, Zhuoran Yang","submitted_at":"2022-02-23T15:27:16Z","abstract_excerpt":"Offline Reinforcement Learning (RL) aims to learn policies from previously collected datasets without exploring the environment. Directly applying off-policy algorithms to offline RL usually fails due to the extrapolation error caused by the out-of-distribution (OOD) actions. Previous methods tackle such problem by penalizing the Q-values of OOD actions or constraining the trained policy to be close to the behavior policy. Nevertheless, such methods typically prevent the generalization of value functions beyond the offline data and also lack precise characterization of OOD data. In this paper,"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2202.11566","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-02-23T15:27:16Z","cross_cats_sorted":[],"title_canon_sha256":"c8f3e10caf03092d3bac029fe7d87051d024acf1bd750980d3b94b3409e848a3","abstract_canon_sha256":"e19ca0de18f33c10f01e62d0ec6f137f647dc1f2cfd9837a51da0a8f34d03d81"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:59:34.499707Z","signature_b64":"V9HwLB29MyEsKgms/QrVDB+JCNGBAzkXQpBeeRGRd4BTZEoqq39jLHT1Y/nOX+MEaJjRAdfsBgcMTdkb0JIvAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4d4604581672df47dd178f489f69614722e2e1361ffa7bd1ef4c1c733c0e86af","last_reissued_at":"2026-07-05T03:59:34.499223Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:59:34.499223Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Pessimistic Bootstrapping for Uncertainty-Driven Offline Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Animesh Garg, Chenjia Bai, Lingxiao Wang, Peng Liu, Zhaoran Wang, Zhihong Deng, Zhuoran Yang","submitted_at":"2022-02-23T15:27:16Z","abstract_excerpt":"Offline Reinforcement Learning (RL) aims to learn policies from previously collected datasets without exploring the environment. Directly applying off-policy algorithms to offline RL usually fails due to the extrapolation error caused by the out-of-distribution (OOD) actions. Previous methods tackle such problem by penalizing the Q-values of OOD actions or constraining the trained policy to be close to the behavior policy. Nevertheless, such methods typically prevent the generalization of value functions beyond the offline data and also lack precise characterization of OOD data. In this paper,"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2202.11566","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2202.11566/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2202.11566","created_at":"2026-07-05T03:59:34.499281+00:00"},{"alias_kind":"arxiv_version","alias_value":"2202.11566v1","created_at":"2026-07-05T03:59:34.499281+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2202.11566","created_at":"2026-07-05T03:59:34.499281+00:00"},{"alias_kind":"pith_short_12","alias_value":"JVDAIWAWOLPU","created_at":"2026-07-05T03:59:34.499281+00:00"},{"alias_kind":"pith_short_16","alias_value":"JVDAIWAWOLPUPXIX","created_at":"2026-07-05T03:59:34.499281+00:00"},{"alias_kind":"pith_short_8","alias_value":"JVDAIWAW","created_at":"2026-07-05T03:59:34.499281+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2504.11944","citing_title":"VIPO: Value Function Inconsistency Penalized Offline Reinforcement Learning","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08202","citing_title":"Beyond Penalization: Diffusion-based Out-of-Distribution Detection and Selective Regularization in Offline Reinforcement Learning","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01862","citing_title":"QHyer: Q-conditioned Hybrid Attention-mamba Transformer for Offline Goal-conditioned RL","ref_index":11,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JVDAIWAWOLPUPXIXR5EJ62LBI4","json":"https://pith.science/pith/JVDAIWAWOLPUPXIXR5EJ62LBI4.json","graph_json":"https://pith.science/api/pith-number/JVDAIWAWOLPUPXIXR5EJ62LBI4/graph.json","events_json":"https://pith.science/api/pith-number/JVDAIWAWOLPUPXIXR5EJ62LBI4/events.json","paper":"https://pith.science/paper/JVDAIWAW"},"agent_actions":{"view_html":"https://pith.science/pith/JVDAIWAWOLPUPXIXR5EJ62LBI4","download_json":"https://pith.science/pith/JVDAIWAWOLPUPXIXR5EJ62LBI4.json","view_paper":"https://pith.science/paper/JVDAIWAW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2202.11566&json=true","fetch_graph":"https://pith.science/api/pith-number/JVDAIWAWOLPUPXIXR5EJ62LBI4/graph.json","fetch_events":"https://pith.science/api/pith-number/JVDAIWAWOLPUPXIXR5EJ62LBI4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JVDAIWAWOLPUPXIXR5EJ62LBI4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JVDAIWAWOLPUPXIXR5EJ62LBI4/action/storage_attestation","attest_author":"https://pith.science/pith/JVDAIWAWOLPUPXIXR5EJ62LBI4/action/author_attestation","sign_citation":"https://pith.science/pith/JVDAIWAWOLPUPXIXR5EJ62LBI4/action/citation_signature","submit_replication":"https://pith.science/pith/JVDAIWAWOLPUPXIXR5EJ62LBI4/action/replication_record"}},"created_at":"2026-07-05T03:59:34.499281+00:00","updated_at":"2026-07-05T03:59:34.499281+00:00"}