{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:CTIQIDZQCSOLMV5RBP4FKV6KYP","short_pith_number":"pith:CTIQIDZQ","schema_version":"1.0","canonical_sha256":"14d1040f30149cb657b10bf85557cac3ffe82d7b0759264b7a249ef6d58bee97","source":{"kind":"arxiv","id":"2211.11802","version":1},"attestation_state":"computed","paper":{"title":"Improving TD3-BC: Relaxed Policy Constraint for Offline Learning and Stable Online Fine-Tuning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Alex Beeson, Giovanni Montana","submitted_at":"2022-11-21T19:10:27Z","abstract_excerpt":"The ability to discover optimal behaviour from fixed data sets has the potential to transfer the successes of reinforcement learning (RL) to domains where data collection is acutely problematic. In this offline setting, a key challenge is overcoming overestimation bias for actions not present in data which, without the ability to correct for via interaction with the environment, can propagate and compound during training, leading to highly sub-optimal policies. One simple method to reduce this bias is to introduce a policy constraint via behavioural cloning (BC), which encourages agents to pic"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2211.11802","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-11-21T19:10:27Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"af7c9995a889bf6f501506fbdb73458511b7ba83fc2671adf2ae6628bd689c31","abstract_canon_sha256":"07320bbb6941e18d320ebbe4a6300f32c330c1ca6e4b617ce45f34c678993a84"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:18:09.580876Z","signature_b64":"WCsAyCZQHLJq4a0Cin5Lv+TGUcew38CzDMMD2UkBgxYkeB9rsbMu8MwSkrcY3QBZG17jwRFsaLoWzETmHGqQAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"14d1040f30149cb657b10bf85557cac3ffe82d7b0759264b7a249ef6d58bee97","last_reissued_at":"2026-07-05T05:18:09.580470Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:18:09.580470Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Improving TD3-BC: Relaxed Policy Constraint for Offline Learning and Stable Online Fine-Tuning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Alex Beeson, Giovanni Montana","submitted_at":"2022-11-21T19:10:27Z","abstract_excerpt":"The ability to discover optimal behaviour from fixed data sets has the potential to transfer the successes of reinforcement learning (RL) to domains where data collection is acutely problematic. In this offline setting, a key challenge is overcoming overestimation bias for actions not present in data which, without the ability to correct for via interaction with the environment, can propagate and compound during training, leading to highly sub-optimal policies. One simple method to reduce this bias is to introduce a policy constraint via behavioural cloning (BC), which encourages agents to pic"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2211.11802","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2211.11802/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2211.11802","created_at":"2026-07-05T05:18:09.580524+00:00"},{"alias_kind":"arxiv_version","alias_value":"2211.11802v1","created_at":"2026-07-05T05:18:09.580524+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2211.11802","created_at":"2026-07-05T05:18:09.580524+00:00"},{"alias_kind":"pith_short_12","alias_value":"CTIQIDZQCSOL","created_at":"2026-07-05T05:18:09.580524+00:00"},{"alias_kind":"pith_short_16","alias_value":"CTIQIDZQCSOLMV5R","created_at":"2026-07-05T05:18:09.580524+00:00"},{"alias_kind":"pith_short_8","alias_value":"CTIQIDZQ","created_at":"2026-07-05T05:18:09.580524+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.11151","citing_title":"RankQ: Offline-to-Online Reinforcement Learning via Self-Supervised Action Ranking","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2602.09580","citing_title":"SERNF: Sample-Efficient Real-World Dexterous Policy Fine-Tuning via Action-Chunked Critics and Normalizing Flows","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11151","citing_title":"RankQ: Offline-to-Online Reinforcement Learning via Self-Supervised Action Ranking","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CTIQIDZQCSOLMV5RBP4FKV6KYP","json":"https://pith.science/pith/CTIQIDZQCSOLMV5RBP4FKV6KYP.json","graph_json":"https://pith.science/api/pith-number/CTIQIDZQCSOLMV5RBP4FKV6KYP/graph.json","events_json":"https://pith.science/api/pith-number/CTIQIDZQCSOLMV5RBP4FKV6KYP/events.json","paper":"https://pith.science/paper/CTIQIDZQ"},"agent_actions":{"view_html":"https://pith.science/pith/CTIQIDZQCSOLMV5RBP4FKV6KYP","download_json":"https://pith.science/pith/CTIQIDZQCSOLMV5RBP4FKV6KYP.json","view_paper":"https://pith.science/paper/CTIQIDZQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2211.11802&json=true","fetch_graph":"https://pith.science/api/pith-number/CTIQIDZQCSOLMV5RBP4FKV6KYP/graph.json","fetch_events":"https://pith.science/api/pith-number/CTIQIDZQCSOLMV5RBP4FKV6KYP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CTIQIDZQCSOLMV5RBP4FKV6KYP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CTIQIDZQCSOLMV5RBP4FKV6KYP/action/storage_attestation","attest_author":"https://pith.science/pith/CTIQIDZQCSOLMV5RBP4FKV6KYP/action/author_attestation","sign_citation":"https://pith.science/pith/CTIQIDZQCSOLMV5RBP4FKV6KYP/action/citation_signature","submit_replication":"https://pith.science/pith/CTIQIDZQCSOLMV5RBP4FKV6KYP/action/replication_record"}},"created_at":"2026-07-05T05:18:09.580524+00:00","updated_at":"2026-07-05T05:18:09.580524+00:00"}