{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:64K7YGDHA4ORUCCTSGAP52XB6X","short_pith_number":"pith:64K7YGDH","schema_version":"1.0","canonical_sha256":"f715fc1867071d1a08539180feeae1f5e6e4e65c53479dbfbf91c1a8ab196b1e","source":{"kind":"arxiv","id":"2002.08396","version":3},"attestation_state":"computed","paper":{"title":"Keep Doing What Worked: Behavioral Modelling Priors for Offline Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.RO","stat.ML"],"primary_cat":"cs.LG","authors_text":"Abbas Abdolmaleki, Felix Berkenkamp, Jost Tobias Springenberg, Martin Riedmiller, Michael Neunert, Nicolas Heess, Noah Y. Siegel, Roland Hafner, Thomas Lampe","submitted_at":"2020-02-19T19:21:08Z","abstract_excerpt":"Off-policy reinforcement learning algorithms promise to be applicable in settings where only a fixed data-set (batch) of environment interactions is available and no new experience can be acquired. This property makes these algorithms appealing for real world problems such as robot control. In practice, however, standard off-policy algorithms fail in the batch setting for continuous control. In this paper, we propose a simple solution to this problem. It admits the use of data generated by arbitrary behavior policies and uses a learned prior -- the advantage-weighted behavior model (ABM) -- to"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2002.08396","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-02-19T19:21:08Z","cross_cats_sorted":["cs.RO","stat.ML"],"title_canon_sha256":"1124434e0f67da539841d259b0ae02b62be845b3e76fe05112738717b6558b68","abstract_canon_sha256":"1a6bfdd94ec5e06f01fef4d520a6b83b3fc611bd0b0effe9373ea4eda24b9a2f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:10:57.948751Z","signature_b64":"QdqRgzXDtTyvaVioxywzHBT9hP0AS4PDhS836vGK56pYbcnhKgAMf26h2UXdVRChAfIcZcfZy+fmVwa0tbpPDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f715fc1867071d1a08539180feeae1f5e6e4e65c53479dbfbf91c1a8ab196b1e","last_reissued_at":"2026-07-05T01:10:57.948220Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:10:57.948220Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Keep Doing What Worked: Behavioral Modelling Priors for Offline Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.RO","stat.ML"],"primary_cat":"cs.LG","authors_text":"Abbas Abdolmaleki, Felix Berkenkamp, Jost Tobias Springenberg, Martin Riedmiller, Michael Neunert, Nicolas Heess, Noah Y. Siegel, Roland Hafner, Thomas Lampe","submitted_at":"2020-02-19T19:21:08Z","abstract_excerpt":"Off-policy reinforcement learning algorithms promise to be applicable in settings where only a fixed data-set (batch) of environment interactions is available and no new experience can be acquired. This property makes these algorithms appealing for real world problems such as robot control. In practice, however, standard off-policy algorithms fail in the batch setting for continuous control. In this paper, we propose a simple solution to this problem. It admits the use of data generated by arbitrary behavior policies and uses a learned prior -- the advantage-weighted behavior model (ABM) -- to"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2002.08396","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2002.08396/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2002.08396","created_at":"2026-07-05T01:10:57.948282+00:00"},{"alias_kind":"arxiv_version","alias_value":"2002.08396v3","created_at":"2026-07-05T01:10:57.948282+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2002.08396","created_at":"2026-07-05T01:10:57.948282+00:00"},{"alias_kind":"pith_short_12","alias_value":"64K7YGDHA4OR","created_at":"2026-07-05T01:10:57.948282+00:00"},{"alias_kind":"pith_short_16","alias_value":"64K7YGDHA4ORUCCT","created_at":"2026-07-05T01:10:57.948282+00:00"},{"alias_kind":"pith_short_8","alias_value":"64K7YGDH","created_at":"2026-07-05T01:10:57.948282+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2601.07060","citing_title":"PALM: Progress-Aware Policy Learning via Affordance Reasoning for Long-Horizon Robotic Manipulation","ref_index":106,"is_internal_anchor":false},{"citing_arxiv_id":"2108.03298","citing_title":"What Matters in Learning from Offline Human Demonstrations for Robot Manipulation","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2005.01643","citing_title":"Offline Reinforcement Learning: Tutorial, Review, and Perspectives on Open Problems","ref_index":179,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02112","citing_title":"An adaptive variance estimator for relative sparsity","ref_index":54,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/64K7YGDHA4ORUCCTSGAP52XB6X","json":"https://pith.science/pith/64K7YGDHA4ORUCCTSGAP52XB6X.json","graph_json":"https://pith.science/api/pith-number/64K7YGDHA4ORUCCTSGAP52XB6X/graph.json","events_json":"https://pith.science/api/pith-number/64K7YGDHA4ORUCCTSGAP52XB6X/events.json","paper":"https://pith.science/paper/64K7YGDH"},"agent_actions":{"view_html":"https://pith.science/pith/64K7YGDHA4ORUCCTSGAP52XB6X","download_json":"https://pith.science/pith/64K7YGDHA4ORUCCTSGAP52XB6X.json","view_paper":"https://pith.science/paper/64K7YGDH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2002.08396&json=true","fetch_graph":"https://pith.science/api/pith-number/64K7YGDHA4ORUCCTSGAP52XB6X/graph.json","fetch_events":"https://pith.science/api/pith-number/64K7YGDHA4ORUCCTSGAP52XB6X/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/64K7YGDHA4ORUCCTSGAP52XB6X/action/timestamp_anchor","attest_storage":"https://pith.science/pith/64K7YGDHA4ORUCCTSGAP52XB6X/action/storage_attestation","attest_author":"https://pith.science/pith/64K7YGDHA4ORUCCTSGAP52XB6X/action/author_attestation","sign_citation":"https://pith.science/pith/64K7YGDHA4ORUCCTSGAP52XB6X/action/citation_signature","submit_replication":"https://pith.science/pith/64K7YGDHA4ORUCCTSGAP52XB6X/action/replication_record"}},"created_at":"2026-07-05T01:10:57.948282+00:00","updated_at":"2026-07-05T01:10:57.948282+00:00"}