{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:67LYCPFQRBKVOF7KKFIK5KJAXS","short_pith_number":"pith:67LYCPFQ","schema_version":"1.0","canonical_sha256":"f7d7813cb088555717ea5150aea920bc939cb194aa957a84860e630277ffa756","source":{"kind":"arxiv","id":"2103.09575","version":1},"attestation_state":"computed","paper":{"title":"Regularized Behavior Value Estimation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Caglar Gulcehre, Jakub Sygnowski, Konrad Zolna, Matthew Hoffman, Nando de Freitas, Razvan Pascanu, Sergio G\\'omez Colmenarejo, Thomas Paine, Yutian Chen, Ziyu Wang","submitted_at":"2021-03-17T11:34:54Z","abstract_excerpt":"Offline reinforcement learning restricts the learning process to rely only on logged-data without access to an environment. While this enables real-world applications, it also poses unique challenges. One important challenge is dealing with errors caused by the overestimation of values for state-action pairs not well-covered by the training data. Due to bootstrapping, these errors get amplified during training and can lead to divergence, thereby crippling learning. To overcome this challenge, we introduce Regularized Behavior Value Estimation (R-BVE). Unlike most approaches, which use policy i"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2103.09575","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-03-17T11:34:54Z","cross_cats_sorted":[],"title_canon_sha256":"fbc004e4dea6bc91341f10ab00914be0280ee23193ec2e98663d38ae8e9d0606","abstract_canon_sha256":"c9075bc558963052913903756c054604507169b0c85e1d640b998b7a127312f7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:24:05.357610Z","signature_b64":"TeEDYtsqNywdcayolkPTJ+y9srLm3+RyDMU3KUZm5AtdwH1yTdoK0JtwRwH+zteXpoqKNAlXK59R4dLnQvoDAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f7d7813cb088555717ea5150aea920bc939cb194aa957a84860e630277ffa756","last_reissued_at":"2026-07-05T02:24:05.357005Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:24:05.357005Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Regularized Behavior Value Estimation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Caglar Gulcehre, Jakub Sygnowski, Konrad Zolna, Matthew Hoffman, Nando de Freitas, Razvan Pascanu, Sergio G\\'omez Colmenarejo, Thomas Paine, Yutian Chen, Ziyu Wang","submitted_at":"2021-03-17T11:34:54Z","abstract_excerpt":"Offline reinforcement learning restricts the learning process to rely only on logged-data without access to an environment. While this enables real-world applications, it also poses unique challenges. One important challenge is dealing with errors caused by the overestimation of values for state-action pairs not well-covered by the training data. Due to bootstrapping, these errors get amplified during training and can lead to divergence, thereby crippling learning. To overcome this challenge, we introduce Regularized Behavior Value Estimation (R-BVE). Unlike most approaches, which use policy i"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2103.09575","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2103.09575/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2103.09575","created_at":"2026-07-05T02:24:05.357081+00:00"},{"alias_kind":"arxiv_version","alias_value":"2103.09575v1","created_at":"2026-07-05T02:24:05.357081+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2103.09575","created_at":"2026-07-05T02:24:05.357081+00:00"},{"alias_kind":"pith_short_12","alias_value":"67LYCPFQRBKV","created_at":"2026-07-05T02:24:05.357081+00:00"},{"alias_kind":"pith_short_16","alias_value":"67LYCPFQRBKVOF7K","created_at":"2026-07-05T02:24:05.357081+00:00"},{"alias_kind":"pith_short_8","alias_value":"67LYCPFQ","created_at":"2026-07-05T02:24:05.357081+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2308.08998","citing_title":"Reinforced Self-Training (ReST) for Language Modeling","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/67LYCPFQRBKVOF7KKFIK5KJAXS","json":"https://pith.science/pith/67LYCPFQRBKVOF7KKFIK5KJAXS.json","graph_json":"https://pith.science/api/pith-number/67LYCPFQRBKVOF7KKFIK5KJAXS/graph.json","events_json":"https://pith.science/api/pith-number/67LYCPFQRBKVOF7KKFIK5KJAXS/events.json","paper":"https://pith.science/paper/67LYCPFQ"},"agent_actions":{"view_html":"https://pith.science/pith/67LYCPFQRBKVOF7KKFIK5KJAXS","download_json":"https://pith.science/pith/67LYCPFQRBKVOF7KKFIK5KJAXS.json","view_paper":"https://pith.science/paper/67LYCPFQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2103.09575&json=true","fetch_graph":"https://pith.science/api/pith-number/67LYCPFQRBKVOF7KKFIK5KJAXS/graph.json","fetch_events":"https://pith.science/api/pith-number/67LYCPFQRBKVOF7KKFIK5KJAXS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/67LYCPFQRBKVOF7KKFIK5KJAXS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/67LYCPFQRBKVOF7KKFIK5KJAXS/action/storage_attestation","attest_author":"https://pith.science/pith/67LYCPFQRBKVOF7KKFIK5KJAXS/action/author_attestation","sign_citation":"https://pith.science/pith/67LYCPFQRBKVOF7KKFIK5KJAXS/action/citation_signature","submit_replication":"https://pith.science/pith/67LYCPFQRBKVOF7KKFIK5KJAXS/action/replication_record"}},"created_at":"2026-07-05T02:24:05.357081+00:00","updated_at":"2026-07-05T02:24:05.357081+00:00"}