{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:HZT6BV3KLFHHR3KX6A2MLUAT7K","short_pith_number":"pith:HZT6BV3K","schema_version":"1.0","canonical_sha256":"3e67e0d76a594e78ed57f034c5d013fa9028dd4b62a299abc00491360493601c","source":{"kind":"arxiv","id":"2106.06860","version":2},"attestation_state":"computed","paper":{"title":"A Minimalist Approach to Offline Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Scott Fujimoto, Shixiang Shane Gu","submitted_at":"2021-06-12T20:38:59Z","abstract_excerpt":"Offline reinforcement learning (RL) defines the task of learning from a fixed batch of data. Due to errors in value estimation from out-of-distribution actions, most offline RL algorithms take the approach of constraining or regularizing the policy with the actions contained in the dataset. Built on pre-existing RL algorithms, modifications to make an RL algorithm work offline comes at the cost of additional complexity. Offline RL algorithms introduce new hyperparameters and often leverage secondary components such as generative models, while adjusting the underlying RL algorithm. In this pape"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2106.06860","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-06-12T20:38:59Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"ee9c395db759ee149a51d37f2f196aa03bfaa1bb6ec7dacb6cd1169b25c1eaad","abstract_canon_sha256":"4d68789136a9024ec1e084e940714be1610af9e9422d3006624c03a8ad1b9d83"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:37:26.368460Z","signature_b64":"SHJ3QCWEu689Gtk7wKxbosd1VOqieWUG4bLVIrckqSCRSY39BrzrxntHklR9kmuxjmZRhqA+ua1u0TKDK7C8DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3e67e0d76a594e78ed57f034c5d013fa9028dd4b62a299abc00491360493601c","last_reissued_at":"2026-07-05T03:37:26.367910Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:37:26.367910Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Minimalist Approach to Offline Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Scott Fujimoto, Shixiang Shane Gu","submitted_at":"2021-06-12T20:38:59Z","abstract_excerpt":"Offline reinforcement learning (RL) defines the task of learning from a fixed batch of data. Due to errors in value estimation from out-of-distribution actions, most offline RL algorithms take the approach of constraining or regularizing the policy with the actions contained in the dataset. Built on pre-existing RL algorithms, modifications to make an RL algorithm work offline comes at the cost of additional complexity. Offline RL algorithms introduce new hyperparameters and often leverage secondary components such as generative models, while adjusting the underlying RL algorithm. In this pape"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2106.06860","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2106.06860/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2106.06860","created_at":"2026-07-05T03:37:26.367968+00:00"},{"alias_kind":"arxiv_version","alias_value":"2106.06860v2","created_at":"2026-07-05T03:37:26.367968+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2106.06860","created_at":"2026-07-05T03:37:26.367968+00:00"},{"alias_kind":"pith_short_12","alias_value":"HZT6BV3KLFHH","created_at":"2026-07-05T03:37:26.367968+00:00"},{"alias_kind":"pith_short_16","alias_value":"HZT6BV3KLFHHR3KX","created_at":"2026-07-05T03:37:26.367968+00:00"},{"alias_kind":"pith_short_8","alias_value":"HZT6BV3K","created_at":"2026-07-05T03:37:26.367968+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.13675","citing_title":"Improving Robotic Generalist Policies via Flow Reversal Steering","ref_index":95,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22711","citing_title":"Abstraction for Offline Goal-Conditioned Reinforcement Learning","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2507.14995","citing_title":"LLM-Enhanced Multi-Agent Reinforcement Learning with Expert Workflow for Real-Time P2P Energy Trading","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13918","citing_title":"CA2: Code-Aware Agent for Automated Game Testing","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2110.06169","citing_title":"Offline Reinforcement Learning with Implicit Q-Learning","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HZT6BV3KLFHHR3KX6A2MLUAT7K","json":"https://pith.science/pith/HZT6BV3KLFHHR3KX6A2MLUAT7K.json","graph_json":"https://pith.science/api/pith-number/HZT6BV3KLFHHR3KX6A2MLUAT7K/graph.json","events_json":"https://pith.science/api/pith-number/HZT6BV3KLFHHR3KX6A2MLUAT7K/events.json","paper":"https://pith.science/paper/HZT6BV3K"},"agent_actions":{"view_html":"https://pith.science/pith/HZT6BV3KLFHHR3KX6A2MLUAT7K","download_json":"https://pith.science/pith/HZT6BV3KLFHHR3KX6A2MLUAT7K.json","view_paper":"https://pith.science/paper/HZT6BV3K","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2106.06860&json=true","fetch_graph":"https://pith.science/api/pith-number/HZT6BV3KLFHHR3KX6A2MLUAT7K/graph.json","fetch_events":"https://pith.science/api/pith-number/HZT6BV3KLFHHR3KX6A2MLUAT7K/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HZT6BV3KLFHHR3KX6A2MLUAT7K/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HZT6BV3KLFHHR3KX6A2MLUAT7K/action/storage_attestation","attest_author":"https://pith.science/pith/HZT6BV3KLFHHR3KX6A2MLUAT7K/action/author_attestation","sign_citation":"https://pith.science/pith/HZT6BV3KLFHHR3KX6A2MLUAT7K/action/citation_signature","submit_replication":"https://pith.science/pith/HZT6BV3KLFHHR3KX6A2MLUAT7K/action/replication_record"}},"created_at":"2026-07-05T03:37:26.367968+00:00","updated_at":"2026-07-05T03:37:26.367968+00:00"}