{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:ZNIOUYAZJ3TSBMN4R67RYEQRV5","short_pith_number":"pith:ZNIOUYAZ","schema_version":"1.0","canonical_sha256":"cb50ea60194ee720b1bc8fbf1c1211af43efefcb2cf2f8c7da71d4957c101246","source":{"kind":"arxiv","id":"2205.13589","version":3},"attestation_state":"computed","paper":{"title":"Pessimism in the Face of Confounders: Provably Efficient Offline Reinforcement Learning in Partially Observable Markov Decision Processes","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","math.ST","stat.ME","stat.ML","stat.TH"],"primary_cat":"cs.LG","authors_text":"Miao Lu, Yifei Min, Zhaoran Wang, Zhuoran Yang","submitted_at":"2022-05-26T19:13:55Z","abstract_excerpt":"We study offline reinforcement learning (RL) in partially observable Markov decision processes. In particular, we aim to learn an optimal policy from a dataset collected by a behavior policy which possibly depends on the latent state. Such a dataset is confounded in the sense that the latent state simultaneously affects the action and the observation, which is prohibitive for existing offline RL algorithms. To this end, we propose the \\underline{P}roxy variable \\underline{P}essimistic \\underline{P}olicy \\underline{O}ptimization (\\texttt{P3O}) algorithm, which addresses the confounding bias and"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2205.13589","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-05-26T19:13:55Z","cross_cats_sorted":["cs.AI","math.ST","stat.ME","stat.ML","stat.TH"],"title_canon_sha256":"cfc120b1e53ce8ed59a8147b87e21476100870eb34ac095c3cd8b9196bc3f342","abstract_canon_sha256":"380e714630780a871cc15b47e0a2f7ca42b9f1fe3d248f51a64e0531ec8825be"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:02:41.153844Z","signature_b64":"1KjWpz6Uv827UqJqXoZFxtjCd7xKJ3MABmfwjmFsBC/V+7Sncm4KWTGJq0nv9kWMvutTc6Z5uZk83oWieZymCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cb50ea60194ee720b1bc8fbf1c1211af43efefcb2cf2f8c7da71d4957c101246","last_reissued_at":"2026-07-05T08:02:41.153422Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:02:41.153422Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Pessimism in the Face of Confounders: Provably Efficient Offline Reinforcement Learning in Partially Observable Markov Decision Processes","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","math.ST","stat.ME","stat.ML","stat.TH"],"primary_cat":"cs.LG","authors_text":"Miao Lu, Yifei Min, Zhaoran Wang, Zhuoran Yang","submitted_at":"2022-05-26T19:13:55Z","abstract_excerpt":"We study offline reinforcement learning (RL) in partially observable Markov decision processes. In particular, we aim to learn an optimal policy from a dataset collected by a behavior policy which possibly depends on the latent state. Such a dataset is confounded in the sense that the latent state simultaneously affects the action and the observation, which is prohibitive for existing offline RL algorithms. To this end, we propose the \\underline{P}roxy variable \\underline{P}essimistic \\underline{P}olicy \\underline{O}ptimization (\\texttt{P3O}) algorithm, which addresses the confounding bias and"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2205.13589","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2205.13589/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2205.13589","created_at":"2026-07-05T08:02:41.153480+00:00"},{"alias_kind":"arxiv_version","alias_value":"2205.13589v3","created_at":"2026-07-05T08:02:41.153480+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2205.13589","created_at":"2026-07-05T08:02:41.153480+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZNIOUYAZJ3TS","created_at":"2026-07-05T08:02:41.153480+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZNIOUYAZJ3TSBMN4","created_at":"2026-07-05T08:02:41.153480+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZNIOUYAZ","created_at":"2026-07-05T08:02:41.153480+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.07140","citing_title":"Quantile-Optimal Policy Learning under Unmeasured Confounding","ref_index":43,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZNIOUYAZJ3TSBMN4R67RYEQRV5","json":"https://pith.science/pith/ZNIOUYAZJ3TSBMN4R67RYEQRV5.json","graph_json":"https://pith.science/api/pith-number/ZNIOUYAZJ3TSBMN4R67RYEQRV5/graph.json","events_json":"https://pith.science/api/pith-number/ZNIOUYAZJ3TSBMN4R67RYEQRV5/events.json","paper":"https://pith.science/paper/ZNIOUYAZ"},"agent_actions":{"view_html":"https://pith.science/pith/ZNIOUYAZJ3TSBMN4R67RYEQRV5","download_json":"https://pith.science/pith/ZNIOUYAZJ3TSBMN4R67RYEQRV5.json","view_paper":"https://pith.science/paper/ZNIOUYAZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2205.13589&json=true","fetch_graph":"https://pith.science/api/pith-number/ZNIOUYAZJ3TSBMN4R67RYEQRV5/graph.json","fetch_events":"https://pith.science/api/pith-number/ZNIOUYAZJ3TSBMN4R67RYEQRV5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZNIOUYAZJ3TSBMN4R67RYEQRV5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZNIOUYAZJ3TSBMN4R67RYEQRV5/action/storage_attestation","attest_author":"https://pith.science/pith/ZNIOUYAZJ3TSBMN4R67RYEQRV5/action/author_attestation","sign_citation":"https://pith.science/pith/ZNIOUYAZJ3TSBMN4R67RYEQRV5/action/citation_signature","submit_replication":"https://pith.science/pith/ZNIOUYAZJ3TSBMN4R67RYEQRV5/action/replication_record"}},"created_at":"2026-07-05T08:02:41.153480+00:00","updated_at":"2026-07-05T08:02:41.153480+00:00"}