{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:G7BDELRRHBXUYLBPLXM4R7R3QW","short_pith_number":"pith:G7BDELRR","schema_version":"1.0","canonical_sha256":"37c2322e31386f4c2c2f5dd9c8fe3b85a9e9e8a8682a60806608863e1ea3216c","source":{"kind":"arxiv","id":"2408.02165","version":1},"attestation_state":"computed","paper":{"title":"SelfBC: Self Behavior Cloning for Offline Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Chenjia Bai, Gaurav Sharma, Hao Zhang, Shirong Liu, Yang Liu, Zixian Guo","submitted_at":"2024-08-04T23:23:48Z","abstract_excerpt":"Policy constraint methods in offline reinforcement learning employ additional regularization techniques to constrain the discrepancy between the learned policy and the offline dataset. However, these methods tend to result in overly conservative policies that resemble the behavior policy, thus limiting their performance. We investigate this limitation and attribute it to the static nature of traditional constraints. In this paper, we propose a novel dynamic policy constraint that restricts the learned policy on the samples generated by the exponential moving average of previously learned polic"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.02165","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-08-04T23:23:48Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"c8a9104b2f700078ad1d20ff4649f99d2b773f0893c6a0349e3e66e3d24c22b0","abstract_canon_sha256":"ab56e409a29781f2a9a73f37b96bbabd0e9245c48c04915fe69f76cbcfc1542f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:52:08.186050Z","signature_b64":"mVgDO1t3p5XPxXROYfOCezuRH+bjPYhpIRmloZmRf9BptjBpj+PUxIvOIFyzUmqbwmHMOwvZI8W+8SZ7x7FzCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"37c2322e31386f4c2c2f5dd9c8fe3b85a9e9e8a8682a60806608863e1ea3216c","last_reissued_at":"2026-07-05T08:52:08.185638Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:52:08.185638Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SelfBC: Self Behavior Cloning for Offline Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Chenjia Bai, Gaurav Sharma, Hao Zhang, Shirong Liu, Yang Liu, Zixian Guo","submitted_at":"2024-08-04T23:23:48Z","abstract_excerpt":"Policy constraint methods in offline reinforcement learning employ additional regularization techniques to constrain the discrepancy between the learned policy and the offline dataset. However, these methods tend to result in overly conservative policies that resemble the behavior policy, thus limiting their performance. We investigate this limitation and attribute it to the static nature of traditional constraints. In this paper, we propose a novel dynamic policy constraint that restricts the learned policy on the samples generated by the exponential moving average of previously learned polic"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.02165","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.02165/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.02165","created_at":"2026-07-05T08:52:08.185693+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.02165v1","created_at":"2026-07-05T08:52:08.185693+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.02165","created_at":"2026-07-05T08:52:08.185693+00:00"},{"alias_kind":"pith_short_12","alias_value":"G7BDELRRHBXU","created_at":"2026-07-05T08:52:08.185693+00:00"},{"alias_kind":"pith_short_16","alias_value":"G7BDELRRHBXUYLBP","created_at":"2026-07-05T08:52:08.185693+00:00"},{"alias_kind":"pith_short_8","alias_value":"G7BDELRR","created_at":"2026-07-05T08:52:08.185693+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.20350","citing_title":"Decision Flow Policy Optimization","ref_index":62,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/G7BDELRRHBXUYLBPLXM4R7R3QW","json":"https://pith.science/pith/G7BDELRRHBXUYLBPLXM4R7R3QW.json","graph_json":"https://pith.science/api/pith-number/G7BDELRRHBXUYLBPLXM4R7R3QW/graph.json","events_json":"https://pith.science/api/pith-number/G7BDELRRHBXUYLBPLXM4R7R3QW/events.json","paper":"https://pith.science/paper/G7BDELRR"},"agent_actions":{"view_html":"https://pith.science/pith/G7BDELRRHBXUYLBPLXM4R7R3QW","download_json":"https://pith.science/pith/G7BDELRRHBXUYLBPLXM4R7R3QW.json","view_paper":"https://pith.science/paper/G7BDELRR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.02165&json=true","fetch_graph":"https://pith.science/api/pith-number/G7BDELRRHBXUYLBPLXM4R7R3QW/graph.json","fetch_events":"https://pith.science/api/pith-number/G7BDELRRHBXUYLBPLXM4R7R3QW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/G7BDELRRHBXUYLBPLXM4R7R3QW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/G7BDELRRHBXUYLBPLXM4R7R3QW/action/storage_attestation","attest_author":"https://pith.science/pith/G7BDELRRHBXUYLBPLXM4R7R3QW/action/author_attestation","sign_citation":"https://pith.science/pith/G7BDELRRHBXUYLBPLXM4R7R3QW/action/citation_signature","submit_replication":"https://pith.science/pith/G7BDELRRHBXUYLBPLXM4R7R3QW/action/replication_record"}},"created_at":"2026-07-05T08:52:08.185693+00:00","updated_at":"2026-07-05T08:52:08.185693+00:00"}