{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:LJYR2RANSRVAEIM52I4ME26CKO","short_pith_number":"pith:LJYR2RAN","schema_version":"1.0","canonical_sha256":"5a711d440d946a02219dd238c26bc2539653cf566f75512c9c7c08458bdabc6a","source":{"kind":"arxiv","id":"2210.06718","version":3},"attestation_state":"computed","paper":{"title":"Hybrid RL: Using Both Offline and Online Data Can Make RL Efficient","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Akshay Krishnamurthy, Ayush Sekhari, J. Andrew Bagnell, Wen Sun, Yifei Zhou, Yuda Song","submitted_at":"2022-10-13T04:19:05Z","abstract_excerpt":"We consider a hybrid reinforcement learning setting (Hybrid RL), in which an agent has access to an offline dataset and the ability to collect experience via real-world online interaction. The framework mitigates the challenges that arise in both pure offline and online RL settings, allowing for the design of simple and highly effective algorithms, in both theory and practice. We demonstrate these advantages by adapting the classical Q learning/iteration algorithm to the hybrid setting, which we call Hybrid Q-Learning or Hy-Q. In our theoretical results, we prove that the algorithm is both com"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2210.06718","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-10-13T04:19:05Z","cross_cats_sorted":[],"title_canon_sha256":"2b4c1f3b9afe5424c93d518c7d2b0dd9c6f1d3236a26ecac497c08b916891e17","abstract_canon_sha256":"e12c3b2602e810db2869e3c644caf259e5b9664d7201f53d205dc97b8d89bca3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:50:09.560881Z","signature_b64":"fMcLDkFOGiRloAjrm4PvtyfxYRagqwtgS/Bt65OP/nGNqmZy4eXGSaXdtjIch4eKu4oq+k5Cav9Hq2bHdA/BBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5a711d440d946a02219dd238c26bc2539653cf566f75512c9c7c08458bdabc6a","last_reissued_at":"2026-07-05T05:50:09.560270Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:50:09.560270Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Hybrid RL: Using Both Offline and Online Data Can Make RL Efficient","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Akshay Krishnamurthy, Ayush Sekhari, J. Andrew Bagnell, Wen Sun, Yifei Zhou, Yuda Song","submitted_at":"2022-10-13T04:19:05Z","abstract_excerpt":"We consider a hybrid reinforcement learning setting (Hybrid RL), in which an agent has access to an offline dataset and the ability to collect experience via real-world online interaction. The framework mitigates the challenges that arise in both pure offline and online RL settings, allowing for the design of simple and highly effective algorithms, in both theory and practice. We demonstrate these advantages by adapting the classical Q learning/iteration algorithm to the hybrid setting, which we call Hybrid Q-Learning or Hy-Q. In our theoretical results, we prove that the algorithm is both com"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2210.06718","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2210.06718/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2210.06718","created_at":"2026-07-05T05:50:09.560346+00:00"},{"alias_kind":"arxiv_version","alias_value":"2210.06718v3","created_at":"2026-07-05T05:50:09.560346+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2210.06718","created_at":"2026-07-05T05:50:09.560346+00:00"},{"alias_kind":"pith_short_12","alias_value":"LJYR2RANSRVA","created_at":"2026-07-05T05:50:09.560346+00:00"},{"alias_kind":"pith_short_16","alias_value":"LJYR2RANSRVAEIM5","created_at":"2026-07-05T05:50:09.560346+00:00"},{"alias_kind":"pith_short_8","alias_value":"LJYR2RAN","created_at":"2026-07-05T05:50:09.560346+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":17,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26006","citing_title":"FORCE: Efficient VLA Reinforcement Fine-Tuning via Value-Calibrated Warm-up and Self-Distillation","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18531","citing_title":"When Does Trajectory-Level Supervision Permit Efficient Offline Reinforcement Learning?","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14779","citing_title":"Peng's Q($\\lambda$) for Conservative Value Estimation in Offline Reinforcement Learning","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00416","citing_title":"Learning While Deploying: Fleet-Scale Reinforcement Learning for Generalist Robot Policies","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05863","citing_title":"SOPE: Stabilizing Off-Policy Evaluation for Online RL with Prior Data","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18675","citing_title":"COOPO: Cyclic Offline-Online Policy Optimization Algorithm","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2507.07986","citing_title":"EXPO: Stable Reinforcement Learning with Expressive Policies","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2510.21060","citing_title":"On the Sample Complexity of Differentially Private Policy Optimization","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10289","citing_title":"Sample-Mean Anchored Thompson Sampling for Offline-to-Online Learning with Distribution Shift","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14497","citing_title":"ROAD: Adaptive Data Mixing for Offline-to-Online Reinforcement Learning via Bi-Level Optimization","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04142","citing_title":"OP-GRPO: Efficient Off-Policy GRPO for Flow-Matching Models","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10289","citing_title":"Sample-Mean Anchored Thompson Sampling for Offline-to-Online Learning with Distribution Shift","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05863","citing_title":"SOPE: Stabilizing Off-Policy Evaluation for Online RL with Prior Data","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00416","citing_title":"Learning While Deploying: Fleet-Scale Reinforcement Learning for Generalist Robot Policies","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08958","citing_title":"WOMBET: World Model-Based Experience Transfer for Robust and Sample-efficient Reinforcement Learning","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13966","citing_title":"Provably Efficient Offline-to-Online Value Adaptation with General Function Approximation","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17919","citing_title":"Fisher Decorator: Refining Flow Policy via a Local Transport Map","ref_index":55,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LJYR2RANSRVAEIM52I4ME26CKO","json":"https://pith.science/pith/LJYR2RANSRVAEIM52I4ME26CKO.json","graph_json":"https://pith.science/api/pith-number/LJYR2RANSRVAEIM52I4ME26CKO/graph.json","events_json":"https://pith.science/api/pith-number/LJYR2RANSRVAEIM52I4ME26CKO/events.json","paper":"https://pith.science/paper/LJYR2RAN"},"agent_actions":{"view_html":"https://pith.science/pith/LJYR2RANSRVAEIM52I4ME26CKO","download_json":"https://pith.science/pith/LJYR2RANSRVAEIM52I4ME26CKO.json","view_paper":"https://pith.science/paper/LJYR2RAN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2210.06718&json=true","fetch_graph":"https://pith.science/api/pith-number/LJYR2RANSRVAEIM52I4ME26CKO/graph.json","fetch_events":"https://pith.science/api/pith-number/LJYR2RANSRVAEIM52I4ME26CKO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LJYR2RANSRVAEIM52I4ME26CKO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LJYR2RANSRVAEIM52I4ME26CKO/action/storage_attestation","attest_author":"https://pith.science/pith/LJYR2RANSRVAEIM52I4ME26CKO/action/author_attestation","sign_citation":"https://pith.science/pith/LJYR2RANSRVAEIM52I4ME26CKO/action/citation_signature","submit_replication":"https://pith.science/pith/LJYR2RANSRVAEIM52I4ME26CKO/action/replication_record"}},"created_at":"2026-07-05T05:50:09.560346+00:00","updated_at":"2026-07-05T05:50:09.560346+00:00"}