{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:2NYSW7EXIRR7KQCQ3NFTVHNBI2","short_pith_number":"pith:2NYSW7EX","schema_version":"1.0","canonical_sha256":"d3712b7c974463f54050db4b3a9da146afb26f8eb6c9dbeeb3b5cbe893e095c2","source":{"kind":"arxiv","id":"2203.05804","version":1},"attestation_state":"computed","paper":{"title":"Near-optimal Offline Reinforcement Learning with Linear Representation: Leveraging Variance Information with Pessimism","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Mengdi Wang, Ming Yin, Yaqi Duan, Yu-Xiang Wang","submitted_at":"2022-03-11T09:00:12Z","abstract_excerpt":"Offline reinforcement learning, which seeks to utilize offline/historical data to optimize sequential decision-making strategies, has gained surging prominence in recent studies. Due to the advantage that appropriate function approximators can help mitigate the sample complexity burden in modern reinforcement learning problems, existing endeavors usually enforce powerful function representation models (e.g. neural networks) to learn the optimal policies. However, a precise understanding of the statistical limits with function representations, remains elusive, even when such a representation is"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2203.05804","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-03-11T09:00:12Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"6f6884f3bf45b16df6e35fc267377176f228d0fbcff9397c7a88ea496ca643b4","abstract_canon_sha256":"2f68f7306bd37f5cde5e59410cb149631cb76ceb226f5e5c19d05aaee1664bdf"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:04:09.783136Z","signature_b64":"1tUFz9+0ze9c8QNP40Kjo20VwIB3E2dwcP+9YheiqRX66SDw9EJDCXicQFlrIJhyod3k/Fl6v4NVjPojzYPxBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d3712b7c974463f54050db4b3a9da146afb26f8eb6c9dbeeb3b5cbe893e095c2","last_reissued_at":"2026-07-05T04:04:09.782621Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:04:09.782621Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Near-optimal Offline Reinforcement Learning with Linear Representation: Leveraging Variance Information with Pessimism","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Mengdi Wang, Ming Yin, Yaqi Duan, Yu-Xiang Wang","submitted_at":"2022-03-11T09:00:12Z","abstract_excerpt":"Offline reinforcement learning, which seeks to utilize offline/historical data to optimize sequential decision-making strategies, has gained surging prominence in recent studies. Due to the advantage that appropriate function approximators can help mitigate the sample complexity burden in modern reinforcement learning problems, existing endeavors usually enforce powerful function representation models (e.g. neural networks) to learn the optimal policies. However, a precise understanding of the statistical limits with function representations, remains elusive, even when such a representation is"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2203.05804","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2203.05804/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2203.05804","created_at":"2026-07-05T04:04:09.782683+00:00"},{"alias_kind":"arxiv_version","alias_value":"2203.05804v1","created_at":"2026-07-05T04:04:09.782683+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2203.05804","created_at":"2026-07-05T04:04:09.782683+00:00"},{"alias_kind":"pith_short_12","alias_value":"2NYSW7EXIRR7","created_at":"2026-07-05T04:04:09.782683+00:00"},{"alias_kind":"pith_short_16","alias_value":"2NYSW7EXIRR7KQCQ","created_at":"2026-07-05T04:04:09.782683+00:00"},{"alias_kind":"pith_short_8","alias_value":"2NYSW7EX","created_at":"2026-07-05T04:04:09.782683+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.18246","citing_title":"Privacy Preserving Reinforcement Learning with One-Sided Feedback","ref_index":96,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13966","citing_title":"Provably Efficient Offline-to-Online Value Adaptation with General Function Approximation","ref_index":11,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2NYSW7EXIRR7KQCQ3NFTVHNBI2","json":"https://pith.science/pith/2NYSW7EXIRR7KQCQ3NFTVHNBI2.json","graph_json":"https://pith.science/api/pith-number/2NYSW7EXIRR7KQCQ3NFTVHNBI2/graph.json","events_json":"https://pith.science/api/pith-number/2NYSW7EXIRR7KQCQ3NFTVHNBI2/events.json","paper":"https://pith.science/paper/2NYSW7EX"},"agent_actions":{"view_html":"https://pith.science/pith/2NYSW7EXIRR7KQCQ3NFTVHNBI2","download_json":"https://pith.science/pith/2NYSW7EXIRR7KQCQ3NFTVHNBI2.json","view_paper":"https://pith.science/paper/2NYSW7EX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2203.05804&json=true","fetch_graph":"https://pith.science/api/pith-number/2NYSW7EXIRR7KQCQ3NFTVHNBI2/graph.json","fetch_events":"https://pith.science/api/pith-number/2NYSW7EXIRR7KQCQ3NFTVHNBI2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2NYSW7EXIRR7KQCQ3NFTVHNBI2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2NYSW7EXIRR7KQCQ3NFTVHNBI2/action/storage_attestation","attest_author":"https://pith.science/pith/2NYSW7EXIRR7KQCQ3NFTVHNBI2/action/author_attestation","sign_citation":"https://pith.science/pith/2NYSW7EXIRR7KQCQ3NFTVHNBI2/action/citation_signature","submit_replication":"https://pith.science/pith/2NYSW7EXIRR7KQCQ3NFTVHNBI2/action/replication_record"}},"created_at":"2026-07-05T04:04:09.782683+00:00","updated_at":"2026-07-05T04:04:09.782683+00:00"}