{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:KZNE6XWTYV7DGEY7EFH335G4SW","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"0152d4fce988298dbcf4651031de9c92feb6de285c83d3af3ecf445cc0a06847","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-02-25T20:07:13Z","title_canon_sha256":"e0517cafa3d473cdb93e2abc846b2229ce03ed478e6a10a15e2ae404f02bf18a"},"schema_version":"1.0","source":{"id":"2402.16181","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2402.16181","created_at":"2026-07-05T07:49:03Z"},{"alias_kind":"arxiv_version","alias_value":"2402.16181v1","created_at":"2026-07-05T07:49:03Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.16181","created_at":"2026-07-05T07:49:03Z"},{"alias_kind":"pith_short_12","alias_value":"KZNE6XWTYV7D","created_at":"2026-07-05T07:49:03Z"},{"alias_kind":"pith_short_16","alias_value":"KZNE6XWTYV7DGEY7","created_at":"2026-07-05T07:49:03Z"},{"alias_kind":"pith_short_8","alias_value":"KZNE6XWT","created_at":"2026-07-05T07:49:03Z"}],"graph_snapshots":[{"event_id":"sha256:0ff10466d3b0fb1a15e0c300d9852a3ed6fa721681f4b4a51e6cf8e207f72372","target":"graph","created_at":"2026-07-05T07:49:03Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2402.16181/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement learning (RL) has become the de facto standard practice for sequential decision-making problems by improving future acting policies with feedback. However, RL algorithms may require extensive trial-and-error interactions to collect useful feedback for improvement. On the other hand, recent developments in large language models (LLMs) have showcased impressive capabilities in language understanding and generation, yet they fall short in exploration and self-improvement capabilities for planning tasks, lacking the ability to autonomously refine their responses based on feedback. Th","authors_text":"Hongxia Yang, Jianbo Yuan, Shenao Zhang, Shuqi Ke, Sirui Zheng, Wanxin Jin, Yingxiang Yang, Zhaoran Wang, Zhihan Liu","cross_cats":["cs.AI"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-02-25T20:07:13Z","title":"How Can LLM Guide RL? A Value-Based Approach"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.16181","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:51301772690f6d1019dfd75b76cc367e49032388706214dac03434c47de85648","target":"record","created_at":"2026-07-05T07:49:03Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"0152d4fce988298dbcf4651031de9c92feb6de285c83d3af3ecf445cc0a06847","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-02-25T20:07:13Z","title_canon_sha256":"e0517cafa3d473cdb93e2abc846b2229ce03ed478e6a10a15e2ae404f02bf18a"},"schema_version":"1.0","source":{"id":"2402.16181","kind":"arxiv","version":1}},"canonical_sha256":"565a4f5ed3c57e33131f214fbdf4dc9582ebc3a3fdec645c602028957e6fe41b","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"565a4f5ed3c57e33131f214fbdf4dc9582ebc3a3fdec645c602028957e6fe41b","first_computed_at":"2026-07-05T07:49:03.557415Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T07:49:03.557415Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"HX8lrJ5B9ER5XjlB3GUo2FQl+ECvNhvOTjdTysyu7LR9R4povHbf9/5cA4Ne7ht3bqpE4kObJcPoS2eSlJQTCQ==","signature_status":"signed_v1","signed_at":"2026-07-05T07:49:03.557898Z","signed_message":"canonical_sha256_bytes"},"source_id":"2402.16181","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:51301772690f6d1019dfd75b76cc367e49032388706214dac03434c47de85648","sha256:0ff10466d3b0fb1a15e0c300d9852a3ed6fa721681f4b4a51e6cf8e207f72372"],"state_sha256":"5093eb8cc66fbcad80b3403e5fb04192925db37900eb3f0fb3c01af106b53633"}