{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:5QMXNH4YQLY2HSZRF2UJYL54HW","short_pith_number":"pith:5QMXNH4Y","schema_version":"1.0","canonical_sha256":"ec19769f9882f1a3cb312ea89c2fbc3d93d0265d73dc2f5c0adc2a84a09cbf14","source":{"kind":"arxiv","id":"2204.05618","version":1},"attestation_state":"computed","paper":{"title":"When Should We Prefer Offline Reinforcement Learning Over Behavioral Cloning?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Anikait Singh, Aviral Kumar, Joey Hong, Sergey Levine","submitted_at":"2022-04-12T08:25:34Z","abstract_excerpt":"Offline reinforcement learning (RL) algorithms can acquire effective policies by utilizing previously collected experience, without any online interaction. It is widely understood that offline RL is able to extract good policies even from highly suboptimal data, a scenario where imitation learning finds suboptimal solutions that do not improve over the demonstrator that generated the dataset. However, another common use case for practitioners is to learn from data that resembles demonstrations. In this case, one can choose to apply offline RL, but can also use behavioral cloning (BC) algorithm"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2204.05618","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-04-12T08:25:34Z","cross_cats_sorted":[],"title_canon_sha256":"30f81c6da18a28d098f0e68b2de10d1ac7302c98e73da5a4df08c4503f1466dd","abstract_canon_sha256":"426daff6eb055232ea91e9fba08134e66531421c51bf7fcfa901a7feed22b4d5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:14:00.275698Z","signature_b64":"1dwTc7Wqk7UfnU5Av68+VTAbVwIZJ5BxhNw4HsYwfZcoI3bgiNOTXNWywECFlBiF1BqqRmUfXajo4z3Cpd8cAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ec19769f9882f1a3cb312ea89c2fbc3d93d0265d73dc2f5c0adc2a84a09cbf14","last_reissued_at":"2026-07-05T04:14:00.275128Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:14:00.275128Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"When Should We Prefer Offline Reinforcement Learning Over Behavioral Cloning?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Anikait Singh, Aviral Kumar, Joey Hong, Sergey Levine","submitted_at":"2022-04-12T08:25:34Z","abstract_excerpt":"Offline reinforcement learning (RL) algorithms can acquire effective policies by utilizing previously collected experience, without any online interaction. It is widely understood that offline RL is able to extract good policies even from highly suboptimal data, a scenario where imitation learning finds suboptimal solutions that do not improve over the demonstrator that generated the dataset. However, another common use case for practitioners is to learn from data that resembles demonstrations. In this case, one can choose to apply offline RL, but can also use behavioral cloning (BC) algorithm"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2204.05618","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2204.05618/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2204.05618","created_at":"2026-07-05T04:14:00.275185+00:00"},{"alias_kind":"arxiv_version","alias_value":"2204.05618v1","created_at":"2026-07-05T04:14:00.275185+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2204.05618","created_at":"2026-07-05T04:14:00.275185+00:00"},{"alias_kind":"pith_short_12","alias_value":"5QMXNH4YQLY2","created_at":"2026-07-05T04:14:00.275185+00:00"},{"alias_kind":"pith_short_16","alias_value":"5QMXNH4YQLY2HSZR","created_at":"2026-07-05T04:14:00.275185+00:00"},{"alias_kind":"pith_short_8","alias_value":"5QMXNH4Y","created_at":"2026-07-05T04:14:00.275185+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24160","citing_title":"An Introduction to Causal Reinforcement Learning","ref_index":145,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27865","citing_title":"From Bootstrapping to Sequence Modeling: A Unified Generative Framework for Personalized Landing-Page Modeling","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2602.06603","citing_title":"The hidden risks of temporal resampling in clinical reinforcement learning","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2210.00030","citing_title":"VIP: Towards Universal Visual Reward and Representation via Value-Implicit Pre-Training","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5QMXNH4YQLY2HSZRF2UJYL54HW","json":"https://pith.science/pith/5QMXNH4YQLY2HSZRF2UJYL54HW.json","graph_json":"https://pith.science/api/pith-number/5QMXNH4YQLY2HSZRF2UJYL54HW/graph.json","events_json":"https://pith.science/api/pith-number/5QMXNH4YQLY2HSZRF2UJYL54HW/events.json","paper":"https://pith.science/paper/5QMXNH4Y"},"agent_actions":{"view_html":"https://pith.science/pith/5QMXNH4YQLY2HSZRF2UJYL54HW","download_json":"https://pith.science/pith/5QMXNH4YQLY2HSZRF2UJYL54HW.json","view_paper":"https://pith.science/paper/5QMXNH4Y","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2204.05618&json=true","fetch_graph":"https://pith.science/api/pith-number/5QMXNH4YQLY2HSZRF2UJYL54HW/graph.json","fetch_events":"https://pith.science/api/pith-number/5QMXNH4YQLY2HSZRF2UJYL54HW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5QMXNH4YQLY2HSZRF2UJYL54HW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5QMXNH4YQLY2HSZRF2UJYL54HW/action/storage_attestation","attest_author":"https://pith.science/pith/5QMXNH4YQLY2HSZRF2UJYL54HW/action/author_attestation","sign_citation":"https://pith.science/pith/5QMXNH4YQLY2HSZRF2UJYL54HW/action/citation_signature","submit_replication":"https://pith.science/pith/5QMXNH4YQLY2HSZRF2UJYL54HW/action/replication_record"}},"created_at":"2026-07-05T04:14:00.275185+00:00","updated_at":"2026-07-05T04:14:00.275185+00:00"}