{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:MJVMJREVM3O4CDAP22UFVAIONX","short_pith_number":"pith:MJVMJREV","schema_version":"1.0","canonical_sha256":"626ac4c49566ddc10c0fd6a85a810e6deacde15d0dd800792b7f3abd92d24316","source":{"kind":"arxiv","id":"2211.04974","version":2},"attestation_state":"computed","paper":{"title":"Leveraging Offline Data in Online Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Aldo Pacchiano, Andrew Wagenmaker","submitted_at":"2022-11-09T15:39:32Z","abstract_excerpt":"Two central paradigms have emerged in the reinforcement learning (RL) community: online RL and offline RL. In the online RL setting, the agent has no prior knowledge of the environment, and must interact with it in order to find an $\\epsilon$-optimal policy. In the offline RL setting, the learner instead has access to a fixed dataset to learn from, but is unable to otherwise interact with the environment, and must obtain the best policy it can from this offline data. Practical scenarios often motivate an intermediate setting: if we have some set of offline data and, in addition, may also inter"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2211.04974","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-11-09T15:39:32Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"40811b7ca27f01e82ab0cbedbe3fbbc8035f0147db2723deae9b8ebf12a97555","abstract_canon_sha256":"8ad6a23866195d469702f654c0c77972cabadc7e103a1c81b515a55100712bbf"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:33:03.181370Z","signature_b64":"uEIGAXMUq6AplBwUiU+yntPCRQxNw9htzwqIn/Lu88ShtayI2dbETTHIOpTflmvjLVYbM9OxCQeeRbQ1BduHBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"626ac4c49566ddc10c0fd6a85a810e6deacde15d0dd800792b7f3abd92d24316","last_reissued_at":"2026-07-05T06:33:03.180881Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:33:03.180881Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Leveraging Offline Data in Online Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Aldo Pacchiano, Andrew Wagenmaker","submitted_at":"2022-11-09T15:39:32Z","abstract_excerpt":"Two central paradigms have emerged in the reinforcement learning (RL) community: online RL and offline RL. In the online RL setting, the agent has no prior knowledge of the environment, and must interact with it in order to find an $\\epsilon$-optimal policy. In the offline RL setting, the learner instead has access to a fixed dataset to learn from, but is unable to otherwise interact with the environment, and must obtain the best policy it can from this offline data. Practical scenarios often motivate an intermediate setting: if we have some set of offline data and, in addition, may also inter"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2211.04974","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2211.04974/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2211.04974","created_at":"2026-07-05T06:33:03.180942+00:00"},{"alias_kind":"arxiv_version","alias_value":"2211.04974v2","created_at":"2026-07-05T06:33:03.180942+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2211.04974","created_at":"2026-07-05T06:33:03.180942+00:00"},{"alias_kind":"pith_short_12","alias_value":"MJVMJREVM3O4","created_at":"2026-07-05T06:33:03.180942+00:00"},{"alias_kind":"pith_short_16","alias_value":"MJVMJREVM3O4CDAP","created_at":"2026-07-05T06:33:03.180942+00:00"},{"alias_kind":"pith_short_8","alias_value":"MJVMJREV","created_at":"2026-07-05T06:33:03.180942+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21483","citing_title":"A note on the convergence guarantees of RLT-based algorithms for polynomial optimization","ref_index":91,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MJVMJREVM3O4CDAP22UFVAIONX","json":"https://pith.science/pith/MJVMJREVM3O4CDAP22UFVAIONX.json","graph_json":"https://pith.science/api/pith-number/MJVMJREVM3O4CDAP22UFVAIONX/graph.json","events_json":"https://pith.science/api/pith-number/MJVMJREVM3O4CDAP22UFVAIONX/events.json","paper":"https://pith.science/paper/MJVMJREV"},"agent_actions":{"view_html":"https://pith.science/pith/MJVMJREVM3O4CDAP22UFVAIONX","download_json":"https://pith.science/pith/MJVMJREVM3O4CDAP22UFVAIONX.json","view_paper":"https://pith.science/paper/MJVMJREV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2211.04974&json=true","fetch_graph":"https://pith.science/api/pith-number/MJVMJREVM3O4CDAP22UFVAIONX/graph.json","fetch_events":"https://pith.science/api/pith-number/MJVMJREVM3O4CDAP22UFVAIONX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MJVMJREVM3O4CDAP22UFVAIONX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MJVMJREVM3O4CDAP22UFVAIONX/action/storage_attestation","attest_author":"https://pith.science/pith/MJVMJREVM3O4CDAP22UFVAIONX/action/author_attestation","sign_citation":"https://pith.science/pith/MJVMJREVM3O4CDAP22UFVAIONX/action/citation_signature","submit_replication":"https://pith.science/pith/MJVMJREVM3O4CDAP22UFVAIONX/action/replication_record"}},"created_at":"2026-07-05T06:33:03.180942+00:00","updated_at":"2026-07-05T06:33:03.180942+00:00"}