{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2019:YI2KVA72CP3U5G4MCNHKZVYG4L","short_pith_number":"pith:YI2KVA72","canonical_record":{"source":{"id":"1911.11361","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-11-26T06:11:34Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"8a1d4cdfef5d2c716945a3a2581f9311a6b8fe0d47dfa6ce2fad30cbfcd81b81","abstract_canon_sha256":"f9f45accd8d913a65cc3758393608e9e673ee338b6190afccfb28737c8de6652"},"schema_version":"1.0"},"canonical_sha256":"c234aa83fa13f74e9b8c134eacd706e2ec1ac572419df9a00bb297e322bce4fb","source":{"kind":"arxiv","id":"1911.11361","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1911.11361","created_at":"2026-07-05T00:22:08Z"},{"alias_kind":"arxiv_version","alias_value":"1911.11361v1","created_at":"2026-07-05T00:22:08Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1911.11361","created_at":"2026-07-05T00:22:08Z"},{"alias_kind":"pith_short_12","alias_value":"YI2KVA72CP3U","created_at":"2026-07-05T00:22:08Z"},{"alias_kind":"pith_short_16","alias_value":"YI2KVA72CP3U5G4M","created_at":"2026-07-05T00:22:08Z"},{"alias_kind":"pith_short_8","alias_value":"YI2KVA72","created_at":"2026-07-05T00:22:08Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2019:YI2KVA72CP3U5G4MCNHKZVYG4L","target":"record","payload":{"canonical_record":{"source":{"id":"1911.11361","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-11-26T06:11:34Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"8a1d4cdfef5d2c716945a3a2581f9311a6b8fe0d47dfa6ce2fad30cbfcd81b81","abstract_canon_sha256":"f9f45accd8d913a65cc3758393608e9e673ee338b6190afccfb28737c8de6652"},"schema_version":"1.0"},"canonical_sha256":"c234aa83fa13f74e9b8c134eacd706e2ec1ac572419df9a00bb297e322bce4fb","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:22:08.920694Z","signature_b64":"b4Unf43zmY1Fi2JP0uCYeGmeGKxyV9VVwQwF6orMiqFMWRYd+JSq7iXexO7/IwtFBa9H0E3lCCTwdTYe0nuNBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c234aa83fa13f74e9b8c134eacd706e2ec1ac572419df9a00bb297e322bce4fb","last_reissued_at":"2026-07-05T00:22:08.920291Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:22:08.920291Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"1911.11361","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T00:22:08Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"A20tKEPDCbpbpj3uCMGJabku8N25y8mu0lgHJSXqZtJiRl5bPGgkO3caMFjbZyAJHnjhpjx3pjFPy+aVZIt6Bw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-06T19:35:59.469888Z"},"content_sha256":"dc85c6c11384c6153cc6e8a76959ff84f69470ea0194bf0436072f05a5659bb2","schema_version":"1.0","event_id":"sha256:dc85c6c11384c6153cc6e8a76959ff84f69470ea0194bf0436072f05a5659bb2"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2019:YI2KVA72CP3U5G4MCNHKZVYG4L","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Behavior Regularized Offline Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"A basic behavior-regularized actor-critic matches complex recent methods on offline continuous control tasks.","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"George Tucker, Ofir Nachum, Yifan Wu","submitted_at":"2019-11-26T06:11:34Z","abstract_excerpt":"In reinforcement learning (RL) research, it is common to assume access to direct online interactions with the environment. However in many real-world applications, access to the environment is limited to a fixed offline dataset of logged experience. In such settings, standard RL algorithms have been shown to diverge or otherwise yield poor performance. Accordingly, recent work has suggested a number of remedies to these issues. In this work, we introduce a general framework, behavior regularized actor critic (BRAC), to empirically evaluate recently proposed methods as well as a number of simpl"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"Surprisingly, we find that many of the technical complexities introduced in recent methods are unnecessary to achieve strong performance.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"The assumption that performance on the chosen continuous control tasks with their specific offline datasets generalizes to broader offline RL settings and that the regularization coefficient can be chosen without introducing hidden overfitting.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"Behavior-regularized actor-critic methods achieve strong offline RL results with simple regularization, rendering many recent technical additions unnecessary.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"A basic behavior-regularized actor-critic matches complex recent methods on offline continuous control tasks.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"791b63df30a119fec5e491a2a662b054218d79b93f13b47c0cef995dc6e4bcf4"},"source":{"id":"1911.11361","kind":"arxiv","version":1},"verdict":{"id":"a768ac76-e650-40bb-b4e3-0ffa5b56358b","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-13T15:14:20.388828Z","strongest_claim":"Surprisingly, we find that many of the technical complexities introduced in recent methods are unnecessary to achieve strong performance.","one_line_summary":"Behavior-regularized actor-critic methods achieve strong offline RL results with simple regularization, rendering many recent technical additions unnecessary.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"The assumption that performance on the chosen continuous control tasks with their specific offline datasets generalizes to broader offline RL settings and that the regularization coefficient can be chosen without introducing hidden overfitting.","pith_extraction_headline":"A basic behavior-regularized actor-critic matches complex recent methods on offline continuous control tasks."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1911.11361/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":20,"sample":[{"doi":"","year":null,"title":"Maximum a Posteriori Policy Optimisation","work_id":"cb0e49f6-4417-4c5e-b4a8-567d7e9e76a7","ref_index":1,"cited_arxiv_id":"1806.06920","is_internal_anchor":false},{"doi":"","year":1907,"title":"Striving for simplicity in off-policy deep reinforcement learning","work_id":"588a4dc4-2821-4960-ab47-4a85390287c3","ref_index":2,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":1995,"title":"Residual algorithms: Reinforcement learning with function approximation","work_id":"e0b56351-3991-4969-8875-4a569f692ccb","ref_index":3,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":null,"title":"Openai gym.arXiv preprint arXiv:1606.01540,","work_id":"6af98f3f-f074-41ae-a689-7dd7b4b8efde","ref_index":4,"cited_arxiv_id":"1606.01540","is_internal_anchor":true},{"doi":"","year":1902,"title":"Diagnosing bottlenecks in deep q-learning algorithms","work_id":"ae65aa93-1aa9-4a49-83a1-3d808bc9a349","ref_index":5,"cited_arxiv_id":"1902.10250","is_internal_anchor":true}],"resolved_work":20,"snapshot_sha256":"e1e281d26cf11c0d5561986661051720939df23df99b308a77823ca00ae25f40","internal_anchors":7},"formal_canon":{"evidence_count":1,"snapshot_sha256":"75a6b9fff07088c754129840cef4441fbc9164d78b142e863b31165e8a0c64c5"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"a768ac76-e650-40bb-b4e3-0ffa5b56358b"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T00:22:08Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"kKmjizr5JfDmunoff37sQuEMr3jf1iWCkmsvIotUCaLhYXn7Zh+pUbGNOWmJNpNBJkJkJBEROfdH1AocPGeeAg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-06T19:35:59.470401Z"},"content_sha256":"f2ddb02307b658368da9ded44653c46835b1fe314ec51606f3d5f7ddc74ca570","schema_version":"1.0","event_id":"sha256:f2ddb02307b658368da9ded44653c46835b1fe314ec51606f3d5f7ddc74ca570"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/YI2KVA72CP3U5G4MCNHKZVYG4L/bundle.json","state_url":"https://pith.science/pith/YI2KVA72CP3U5G4MCNHKZVYG4L/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/YI2KVA72CP3U5G4MCNHKZVYG4L/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-06T19:35:59Z","links":{"resolver":"https://pith.science/pith/YI2KVA72CP3U5G4MCNHKZVYG4L","bundle":"https://pith.science/pith/YI2KVA72CP3U5G4MCNHKZVYG4L/bundle.json","state":"https://pith.science/pith/YI2KVA72CP3U5G4MCNHKZVYG4L/state.json","well_known_bundle":"https://pith.science/.well-known/pith/YI2KVA72CP3U5G4MCNHKZVYG4L/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2019:YI2KVA72CP3U5G4MCNHKZVYG4L","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"f9f45accd8d913a65cc3758393608e9e673ee338b6190afccfb28737c8de6652","cross_cats_sorted":["cs.AI","stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-11-26T06:11:34Z","title_canon_sha256":"8a1d4cdfef5d2c716945a3a2581f9311a6b8fe0d47dfa6ce2fad30cbfcd81b81"},"schema_version":"1.0","source":{"id":"1911.11361","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1911.11361","created_at":"2026-07-05T00:22:08Z"},{"alias_kind":"arxiv_version","alias_value":"1911.11361v1","created_at":"2026-07-05T00:22:08Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1911.11361","created_at":"2026-07-05T00:22:08Z"},{"alias_kind":"pith_short_12","alias_value":"YI2KVA72CP3U","created_at":"2026-07-05T00:22:08Z"},{"alias_kind":"pith_short_16","alias_value":"YI2KVA72CP3U5G4M","created_at":"2026-07-05T00:22:08Z"},{"alias_kind":"pith_short_8","alias_value":"YI2KVA72","created_at":"2026-07-05T00:22:08Z"}],"graph_snapshots":[{"event_id":"sha256:f2ddb02307b658368da9ded44653c46835b1fe314ec51606f3d5f7ddc74ca570","target":"graph","created_at":"2026-07-05T00:22:08Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"Surprisingly, we find that many of the technical complexities introduced in recent methods are unnecessary to achieve strong performance."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"The assumption that performance on the chosen continuous control tasks with their specific offline datasets generalizes to broader offline RL settings and that the regularization coefficient can be chosen without introducing hidden overfitting."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"Behavior-regularized actor-critic methods achieve strong offline RL results with simple regularization, rendering many recent technical additions unnecessary."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"A basic behavior-regularized actor-critic matches complex recent methods on offline continuous control tasks."}],"snapshot_sha256":"791b63df30a119fec5e491a2a662b054218d79b93f13b47c0cef995dc6e4bcf4"},"formal_canon":{"evidence_count":1,"snapshot_sha256":"75a6b9fff07088c754129840cef4441fbc9164d78b142e863b31165e8a0c64c5"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/1911.11361/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"In reinforcement learning (RL) research, it is common to assume access to direct online interactions with the environment. However in many real-world applications, access to the environment is limited to a fixed offline dataset of logged experience. In such settings, standard RL algorithms have been shown to diverge or otherwise yield poor performance. Accordingly, recent work has suggested a number of remedies to these issues. In this work, we introduce a general framework, behavior regularized actor critic (BRAC), to empirically evaluate recently proposed methods as well as a number of simpl","authors_text":"George Tucker, Ofir Nachum, Yifan Wu","cross_cats":["cs.AI","stat.ML"],"headline":"A basic behavior-regularized actor-critic matches complex recent methods on offline continuous control tasks.","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-11-26T06:11:34Z","title":"Behavior Regularized Offline Reinforcement Learning"},"references":{"count":20,"internal_anchors":7,"resolved_work":20,"sample":[{"cited_arxiv_id":"1806.06920","doi":"","is_internal_anchor":false,"ref_index":1,"title":"Maximum a Posteriori Policy Optimisation","work_id":"cb0e49f6-4417-4c5e-b4a8-567d7e9e76a7","year":null},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":2,"title":"Striving for simplicity in off-policy deep reinforcement learning","work_id":"588a4dc4-2821-4960-ab47-4a85390287c3","year":1907},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":3,"title":"Residual algorithms: Reinforcement learning with function approximation","work_id":"e0b56351-3991-4969-8875-4a569f692ccb","year":1995},{"cited_arxiv_id":"1606.01540","doi":"","is_internal_anchor":true,"ref_index":4,"title":"Openai gym.arXiv preprint arXiv:1606.01540,","work_id":"6af98f3f-f074-41ae-a689-7dd7b4b8efde","year":null},{"cited_arxiv_id":"1902.10250","doi":"","is_internal_anchor":true,"ref_index":5,"title":"Diagnosing bottlenecks in deep q-learning algorithms","work_id":"ae65aa93-1aa9-4a49-83a1-3d808bc9a349","year":1902}],"snapshot_sha256":"e1e281d26cf11c0d5561986661051720939df23df99b308a77823ca00ae25f40"},"source":{"id":"1911.11361","kind":"arxiv","version":1},"verdict":{"created_at":"2026-05-13T15:14:20.388828Z","id":"a768ac76-e650-40bb-b4e3-0ffa5b56358b","model_set":{"reader":"grok-4.3"},"one_line_summary":"Behavior-regularized actor-critic methods achieve strong offline RL results with simple regularization, rendering many recent technical additions unnecessary.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"A basic behavior-regularized actor-critic matches complex recent methods on offline continuous control tasks.","strongest_claim":"Surprisingly, we find that many of the technical complexities introduced in recent methods are unnecessary to achieve strong performance.","weakest_assumption":"The assumption that performance on the chosen continuous control tasks with their specific offline datasets generalizes to broader offline RL settings and that the regularization coefficient can be chosen without introducing hidden overfitting."}},"verdict_id":"a768ac76-e650-40bb-b4e3-0ffa5b56358b"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:dc85c6c11384c6153cc6e8a76959ff84f69470ea0194bf0436072f05a5659bb2","target":"record","created_at":"2026-07-05T00:22:08Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"f9f45accd8d913a65cc3758393608e9e673ee338b6190afccfb28737c8de6652","cross_cats_sorted":["cs.AI","stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-11-26T06:11:34Z","title_canon_sha256":"8a1d4cdfef5d2c716945a3a2581f9311a6b8fe0d47dfa6ce2fad30cbfcd81b81"},"schema_version":"1.0","source":{"id":"1911.11361","kind":"arxiv","version":1}},"canonical_sha256":"c234aa83fa13f74e9b8c134eacd706e2ec1ac572419df9a00bb297e322bce4fb","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"c234aa83fa13f74e9b8c134eacd706e2ec1ac572419df9a00bb297e322bce4fb","first_computed_at":"2026-07-05T00:22:08.920291Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T00:22:08.920291Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"b4Unf43zmY1Fi2JP0uCYeGmeGKxyV9VVwQwF6orMiqFMWRYd+JSq7iXexO7/IwtFBa9H0E3lCCTwdTYe0nuNBw==","signature_status":"signed_v1","signed_at":"2026-07-05T00:22:08.920694Z","signed_message":"canonical_sha256_bytes"},"source_id":"1911.11361","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:dc85c6c11384c6153cc6e8a76959ff84f69470ea0194bf0436072f05a5659bb2","sha256:f2ddb02307b658368da9ded44653c46835b1fe314ec51606f3d5f7ddc74ca570"],"state_sha256":"aee11d4361c3c63e94279b312ca685f92da593a788e7fc76e979d1ba4d2e68d8"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"tvY4JPXA4trdYNBxhNbGj6o6tPWOHyiviJ9q+BQ1+8qNgfGFquhQxS56K0lnMx6fNhMy9YhepTsQzovjubs5DQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-06T19:35:59.473715Z","bundle_sha256":"758c7e490e4e2d3775ff3562c3ba0f617f79a511b8eb1f88fd449e261d54a7ae"}}