{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:AJCFPMCGFE2SNRQR3SLB6JQIBE","short_pith_number":"pith:AJCFPMCG","schema_version":"1.0","canonical_sha256":"024457b046293526c611dc961f2608091ca5418dc669982c069243fec4f46468","source":{"kind":"arxiv","id":"2007.11091","version":2},"attestation_state":"computed","paper":{"title":"EMaQ: Expected-Max Q-Learning Operator for Simple Yet Effective Offline and Online RL","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Dale Schuurmans, Seyed Kamyar Seyed Ghasemipour, Shixiang Shane Gu","submitted_at":"2020-07-21T21:13:02Z","abstract_excerpt":"Off-policy reinforcement learning holds the promise of sample-efficient learning of decision-making policies by leveraging past experience. However, in the offline RL setting -- where a fixed collection of interactions are provided and no further interactions are allowed -- it has been shown that standard off-policy RL methods can significantly underperform. Recently proposed methods often aim to address this shortcoming by constraining learned policies to remain close to the given dataset of interactions. In this work, we closely investigate an important simplification of BCQ -- a prior appro"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2007.11091","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-07-21T21:13:02Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"39e4a24989ff05f94573e11758bc99aa131f93d32b12b2f0d156e66c8b0403ba","abstract_canon_sha256":"8bfec5abda58150a8fcf4f252a4231d49201fcb1b50c32f8160b8a7f0998c832"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:06:51.146433Z","signature_b64":"QWYjSHDr7ez0duSkIzLN8a1oldAgWctajTLtoZm2xXG93sJA2fveV0rn4T/Kaj2w2oDpVbvc9W7oeXKSEB7EAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"024457b046293526c611dc961f2608091ca5418dc669982c069243fec4f46468","last_reissued_at":"2026-07-05T02:06:51.145867Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:06:51.145867Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"EMaQ: Expected-Max Q-Learning Operator for Simple Yet Effective Offline and Online RL","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Dale Schuurmans, Seyed Kamyar Seyed Ghasemipour, Shixiang Shane Gu","submitted_at":"2020-07-21T21:13:02Z","abstract_excerpt":"Off-policy reinforcement learning holds the promise of sample-efficient learning of decision-making policies by leveraging past experience. However, in the offline RL setting -- where a fixed collection of interactions are provided and no further interactions are allowed -- it has been shown that standard off-policy RL methods can significantly underperform. Recently proposed methods often aim to address this shortcoming by constraining learned policies to remain close to the given dataset of interactions. In this work, we closely investigate an important simplification of BCQ -- a prior appro"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2007.11091","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2007.11091/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2007.11091","created_at":"2026-07-05T02:06:51.145941+00:00"},{"alias_kind":"arxiv_version","alias_value":"2007.11091v2","created_at":"2026-07-05T02:06:51.145941+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2007.11091","created_at":"2026-07-05T02:06:51.145941+00:00"},{"alias_kind":"pith_short_12","alias_value":"AJCFPMCGFE2S","created_at":"2026-07-05T02:06:51.145941+00:00"},{"alias_kind":"pith_short_16","alias_value":"AJCFPMCGFE2SNRQR","created_at":"2026-07-05T02:06:51.145941+00:00"},{"alias_kind":"pith_short_8","alias_value":"AJCFPMCG","created_at":"2026-07-05T02:06:51.145941+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.22711","citing_title":"Abstraction for Offline Goal-Conditioned Reinforcement Learning","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2108.03298","citing_title":"What Matters in Learning from Offline Human Demonstrations for Robot Manipulation","ref_index":42,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AJCFPMCGFE2SNRQR3SLB6JQIBE","json":"https://pith.science/pith/AJCFPMCGFE2SNRQR3SLB6JQIBE.json","graph_json":"https://pith.science/api/pith-number/AJCFPMCGFE2SNRQR3SLB6JQIBE/graph.json","events_json":"https://pith.science/api/pith-number/AJCFPMCGFE2SNRQR3SLB6JQIBE/events.json","paper":"https://pith.science/paper/AJCFPMCG"},"agent_actions":{"view_html":"https://pith.science/pith/AJCFPMCGFE2SNRQR3SLB6JQIBE","download_json":"https://pith.science/pith/AJCFPMCGFE2SNRQR3SLB6JQIBE.json","view_paper":"https://pith.science/paper/AJCFPMCG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2007.11091&json=true","fetch_graph":"https://pith.science/api/pith-number/AJCFPMCGFE2SNRQR3SLB6JQIBE/graph.json","fetch_events":"https://pith.science/api/pith-number/AJCFPMCGFE2SNRQR3SLB6JQIBE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AJCFPMCGFE2SNRQR3SLB6JQIBE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AJCFPMCGFE2SNRQR3SLB6JQIBE/action/storage_attestation","attest_author":"https://pith.science/pith/AJCFPMCGFE2SNRQR3SLB6JQIBE/action/author_attestation","sign_citation":"https://pith.science/pith/AJCFPMCGFE2SNRQR3SLB6JQIBE/action/citation_signature","submit_replication":"https://pith.science/pith/AJCFPMCGFE2SNRQR3SLB6JQIBE/action/replication_record"}},"created_at":"2026-07-05T02:06:51.145941+00:00","updated_at":"2026-07-05T02:06:51.145941+00:00"}