{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:4SMMISY54NVSQTANBC4J237YZO","short_pith_number":"pith:4SMMISY5","schema_version":"1.0","canonical_sha256":"e498c44b1de36b284c0d08b89d6ff8cbb41df818105733f28f7d1aca7b2b0435","source":{"kind":"arxiv","id":"2102.01748","version":1},"attestation_state":"computed","paper":{"title":"Near-Optimal Offline Reinforcement Learning via Double Variance Reduction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Ming Yin, Yu Bai, Yu-Xiang Wang","submitted_at":"2021-02-02T20:47:35Z","abstract_excerpt":"We consider the problem of offline reinforcement learning (RL) -- a well-motivated setting of RL that aims at policy optimization using only historical data. Despite its wide applicability, theoretical understandings of offline RL, such as its optimal sample complexity, remain largely open even in basic settings such as \\emph{tabular} Markov Decision Processes (MDPs).\n  In this paper, we propose Off-Policy Double Variance Reduction (OPDVR), a new variance reduction based algorithm for offline RL. Our main result shows that OPDVR provably identifies an $\\epsilon$-optimal policy with $\\widetilde"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2102.01748","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-02-02T20:47:35Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"9761d5c48b98597fae39b478118003366210c13b921433e5e602744cceb0a2a4","abstract_canon_sha256":"e87c6914e50eda567682ab0c6a1f1798f0cfc08531560ef4fb6622a941e1858f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:12:33.035968Z","signature_b64":"wb2OOF36swqmfTSzIdiyoz881Z5UrDq9K/h/o4IsEwSnK4Y8+c3/hcWcMwv5a/QrsUg34/vWFePOQHUNwSUSBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e498c44b1de36b284c0d08b89d6ff8cbb41df818105733f28f7d1aca7b2b0435","last_reissued_at":"2026-07-05T02:12:33.035619Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:12:33.035619Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Near-Optimal Offline Reinforcement Learning via Double Variance Reduction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Ming Yin, Yu Bai, Yu-Xiang Wang","submitted_at":"2021-02-02T20:47:35Z","abstract_excerpt":"We consider the problem of offline reinforcement learning (RL) -- a well-motivated setting of RL that aims at policy optimization using only historical data. Despite its wide applicability, theoretical understandings of offline RL, such as its optimal sample complexity, remain largely open even in basic settings such as \\emph{tabular} Markov Decision Processes (MDPs).\n  In this paper, we propose Off-Policy Double Variance Reduction (OPDVR), a new variance reduction based algorithm for offline RL. Our main result shows that OPDVR provably identifies an $\\epsilon$-optimal policy with $\\widetilde"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2102.01748","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2102.01748/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2102.01748","created_at":"2026-07-05T02:12:33.035675+00:00"},{"alias_kind":"arxiv_version","alias_value":"2102.01748v1","created_at":"2026-07-05T02:12:33.035675+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2102.01748","created_at":"2026-07-05T02:12:33.035675+00:00"},{"alias_kind":"pith_short_12","alias_value":"4SMMISY54NVS","created_at":"2026-07-05T02:12:33.035675+00:00"},{"alias_kind":"pith_short_16","alias_value":"4SMMISY54NVSQTAN","created_at":"2026-07-05T02:12:33.035675+00:00"},{"alias_kind":"pith_short_8","alias_value":"4SMMISY5","created_at":"2026-07-05T02:12:33.035675+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2211.15657","citing_title":"Is Conditional Generative Modeling all you need for Decision-Making?","ref_index":189,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4SMMISY54NVSQTANBC4J237YZO","json":"https://pith.science/pith/4SMMISY54NVSQTANBC4J237YZO.json","graph_json":"https://pith.science/api/pith-number/4SMMISY54NVSQTANBC4J237YZO/graph.json","events_json":"https://pith.science/api/pith-number/4SMMISY54NVSQTANBC4J237YZO/events.json","paper":"https://pith.science/paper/4SMMISY5"},"agent_actions":{"view_html":"https://pith.science/pith/4SMMISY54NVSQTANBC4J237YZO","download_json":"https://pith.science/pith/4SMMISY54NVSQTANBC4J237YZO.json","view_paper":"https://pith.science/paper/4SMMISY5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2102.01748&json=true","fetch_graph":"https://pith.science/api/pith-number/4SMMISY54NVSQTANBC4J237YZO/graph.json","fetch_events":"https://pith.science/api/pith-number/4SMMISY54NVSQTANBC4J237YZO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4SMMISY54NVSQTANBC4J237YZO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4SMMISY54NVSQTANBC4J237YZO/action/storage_attestation","attest_author":"https://pith.science/pith/4SMMISY54NVSQTANBC4J237YZO/action/author_attestation","sign_citation":"https://pith.science/pith/4SMMISY54NVSQTANBC4J237YZO/action/citation_signature","submit_replication":"https://pith.science/pith/4SMMISY54NVSQTANBC4J237YZO/action/replication_record"}},"created_at":"2026-07-05T02:12:33.035675+00:00","updated_at":"2026-07-05T02:12:33.035675+00:00"}