{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:RQ2AC355JH2IR2HF7YOGOL6FCS","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"8abb2d05e5af6a010b8e7633cb06f477d5d33a8cd4891e51ba3d1b9d31d8aff3","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-05-23T14:53:54Z","title_canon_sha256":"2b85a3456dff451dd108feb3442e5f9f1567ff161086a03923b9afde920b263b"},"schema_version":"1.0","source":{"id":"2405.14655","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2405.14655","created_at":"2026-07-05T09:43:07Z"},{"alias_kind":"arxiv_version","alias_value":"2405.14655v2","created_at":"2026-07-05T09:43:07Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.14655","created_at":"2026-07-05T09:43:07Z"},{"alias_kind":"pith_short_12","alias_value":"RQ2AC355JH2I","created_at":"2026-07-05T09:43:07Z"},{"alias_kind":"pith_short_16","alias_value":"RQ2AC355JH2IR2HF","created_at":"2026-07-05T09:43:07Z"},{"alias_kind":"pith_short_8","alias_value":"RQ2AC355","created_at":"2026-07-05T09:43:07Z"}],"graph_snapshots":[{"event_id":"sha256:7043f98cb5ce5f996909616652d12843d1ac3453c46012b8a9dba4c04c554551","target":"graph","created_at":"2026-07-05T09:43:07Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2405.14655/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement Learning from Human Feedback (RLHF) has become the standard approach for aligning Large Language Models (LLMs) with human preferences, allowing LLMs to demonstrate remarkable abilities in various tasks. Existing methods work by emulating the preferences at the single decision (turn) level, limiting their capabilities in settings that require planning or multi-turn interactions to achieve a long-term goal. In this paper, we address this issue by developing novel methods for Reinforcement Learning (RL) from preference feedback between two full multi-turn conversations. In the tabul","authors_text":"Asaf Cassel, Avinatan Hassidim, Avital Zipori, Aviv Rosenberg, Bilal Piot, Daniele Calandriello, Hila Noga, Idan Szpektor, Lior Shani, Oran Lang, Orgad Keller, R\\'emi Munos, Yossi Matias","cross_cats":[],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-05-23T14:53:54Z","title":"Multi-turn Reinforcement Learning from Preference Human Feedback"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.14655","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:efa033d400ff99e49ac0e252946ef146b3497fdbac23e8cb76d246bfdfcd96db","target":"record","created_at":"2026-07-05T09:43:07Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"8abb2d05e5af6a010b8e7633cb06f477d5d33a8cd4891e51ba3d1b9d31d8aff3","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-05-23T14:53:54Z","title_canon_sha256":"2b85a3456dff451dd108feb3442e5f9f1567ff161086a03923b9afde920b263b"},"schema_version":"1.0","source":{"id":"2405.14655","kind":"arxiv","version":2}},"canonical_sha256":"8c34016fbd49f488e8e5fe1c672fc5148db70033f17fbe9b31bde71a9e217e60","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"8c34016fbd49f488e8e5fe1c672fc5148db70033f17fbe9b31bde71a9e217e60","first_computed_at":"2026-07-05T09:43:07.280550Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T09:43:07.280550Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"tfzBhaGVIV8HDAOeI8YcoZtv1Zi4vQPgJaPAVBuDvQQnb+Iu8gihL6Tp/9wA5h4xn8hx5cxhOd+/3+wH9cunBw==","signature_status":"signed_v1","signed_at":"2026-07-05T09:43:07.282422Z","signed_message":"canonical_sha256_bytes"},"source_id":"2405.14655","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:efa033d400ff99e49ac0e252946ef146b3497fdbac23e8cb76d246bfdfcd96db","sha256:7043f98cb5ce5f996909616652d12843d1ac3453c46012b8a9dba4c04c554551"],"state_sha256":"d60bbf04ce1a74bf644f694bb7830a144e565ff4ed198301c6b18838c31b1a32"}