{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2023:KFKLSHWLQ6HGIO5AWDCJGXONLV","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"6c739716ce0e5556e5e8edc100c80decd14470213bee567c17fff4a9448460b7","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2023-07-24T16:52:31Z","title_canon_sha256":"acd41d47b77e0941427673e8a6776eede6bda0d175ae55d040f50a51de7da044"},"schema_version":"1.0","source":{"id":"2307.12933","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2307.12933","created_at":"2026-07-05T06:34:00Z"},{"alias_kind":"arxiv_version","alias_value":"2307.12933v1","created_at":"2026-07-05T06:34:00Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.12933","created_at":"2026-07-05T06:34:00Z"},{"alias_kind":"pith_short_12","alias_value":"KFKLSHWLQ6HG","created_at":"2026-07-05T06:34:00Z"},{"alias_kind":"pith_short_16","alias_value":"KFKLSHWLQ6HGIO5A","created_at":"2026-07-05T06:34:00Z"},{"alias_kind":"pith_short_8","alias_value":"KFKLSHWL","created_at":"2026-07-05T06:34:00Z"}],"graph_snapshots":[{"event_id":"sha256:dcc4d2d397cfeedff70eb3a95604ecfb7474700609ef31bc9529ec31166fd816","target":"graph","created_at":"2026-07-05T06:34:00Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2307.12933/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Model-based reinforcement learning (RL) has demonstrated remarkable successes on a range of continuous control tasks due to its high sample efficiency. To save the computation cost of conducting planning online, recent practices tend to distill optimized action sequences into an RL policy during the training phase. Although the distillation can incorporate both the foresight of planning and the exploration ability of RL policies, the theoretical understanding of these methods is yet unclear. In this paper, we extend the policy improvement step of Soft Actor-Critic (SAC) by developing an approa","authors_text":"Chuming Li, Jie Liu, Ruonan Jia, Wanli Ouyang, Yaodong Yang, Yazhe Niu, Yinmin Zhang, Yu Liu","cross_cats":[],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2023-07-24T16:52:31Z","title":"Theoretically Guaranteed Policy Improvement Distilled from Model-Based Planning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.12933","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:2b54c27cc883125e70e25fc8ca319c62b8ee8d7977140a29a6756422b8e7e4d3","target":"record","created_at":"2026-07-05T06:34:00Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"6c739716ce0e5556e5e8edc100c80decd14470213bee567c17fff4a9448460b7","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2023-07-24T16:52:31Z","title_canon_sha256":"acd41d47b77e0941427673e8a6776eede6bda0d175ae55d040f50a51de7da044"},"schema_version":"1.0","source":{"id":"2307.12933","kind":"arxiv","version":1}},"canonical_sha256":"5154b91ecb878e643ba0b0c4935dcd5d54b448020e16e2829bcb00f70afb67b7","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"5154b91ecb878e643ba0b0c4935dcd5d54b448020e16e2829bcb00f70afb67b7","first_computed_at":"2026-07-05T06:34:00.201918Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T06:34:00.201918Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"KPAu+xZYwpHvM/HFYfxMsa1ghUalnVxJ51T47yrd/bpPU8HSbfDDYq+8u7fb1A0p+rFdp4P85E5O3d/Tb1chAQ==","signature_status":"signed_v1","signed_at":"2026-07-05T06:34:00.202364Z","signed_message":"canonical_sha256_bytes"},"source_id":"2307.12933","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:2b54c27cc883125e70e25fc8ca319c62b8ee8d7977140a29a6756422b8e7e4d3","sha256:dcc4d2d397cfeedff70eb3a95604ecfb7474700609ef31bc9529ec31166fd816"],"state_sha256":"c778e32ddd307dee1402aaa31c460c2d03f3c580d87a0453dfb715c26b9f9bdd"}