{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:RQ2AC355JH2IR2HF7YOGOL6FCS","short_pith_number":"pith:RQ2AC355","canonical_record":{"source":{"id":"2405.14655","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-05-23T14:53:54Z","cross_cats_sorted":[],"title_canon_sha256":"2b85a3456dff451dd108feb3442e5f9f1567ff161086a03923b9afde920b263b","abstract_canon_sha256":"8abb2d05e5af6a010b8e7633cb06f477d5d33a8cd4891e51ba3d1b9d31d8aff3"},"schema_version":"1.0"},"canonical_sha256":"8c34016fbd49f488e8e5fe1c672fc5148db70033f17fbe9b31bde71a9e217e60","source":{"kind":"arxiv","id":"2405.14655","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2405.14655","created_at":"2026-07-05T09:43:07Z"},{"alias_kind":"arxiv_version","alias_value":"2405.14655v2","created_at":"2026-07-05T09:43:07Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.14655","created_at":"2026-07-05T09:43:07Z"},{"alias_kind":"pith_short_12","alias_value":"RQ2AC355JH2I","created_at":"2026-07-05T09:43:07Z"},{"alias_kind":"pith_short_16","alias_value":"RQ2AC355JH2IR2HF","created_at":"2026-07-05T09:43:07Z"},{"alias_kind":"pith_short_8","alias_value":"RQ2AC355","created_at":"2026-07-05T09:43:07Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:RQ2AC355JH2IR2HF7YOGOL6FCS","target":"record","payload":{"canonical_record":{"source":{"id":"2405.14655","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-05-23T14:53:54Z","cross_cats_sorted":[],"title_canon_sha256":"2b85a3456dff451dd108feb3442e5f9f1567ff161086a03923b9afde920b263b","abstract_canon_sha256":"8abb2d05e5af6a010b8e7633cb06f477d5d33a8cd4891e51ba3d1b9d31d8aff3"},"schema_version":"1.0"},"canonical_sha256":"8c34016fbd49f488e8e5fe1c672fc5148db70033f17fbe9b31bde71a9e217e60","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:43:07.282422Z","signature_b64":"tfzBhaGVIV8HDAOeI8YcoZtv1Zi4vQPgJaPAVBuDvQQnb+Iu8gihL6Tp/9wA5h4xn8hx5cxhOd+/3+wH9cunBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8c34016fbd49f488e8e5fe1c672fc5148db70033f17fbe9b31bde71a9e217e60","last_reissued_at":"2026-07-05T09:43:07.280550Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:43:07.280550Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2405.14655","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:43:07Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"kmByLHhJhe2RutFSFNRskoBwNCHfcWn/lID8X6Ar/IX0Q+sZk+QMQlveiTLoWPEFjyg/ehBRa2sMOU9ODNDpAw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-08T08:49:34.567351Z"},"content_sha256":"efa033d400ff99e49ac0e252946ef146b3497fdbac23e8cb76d246bfdfcd96db","schema_version":"1.0","event_id":"sha256:efa033d400ff99e49ac0e252946ef146b3497fdbac23e8cb76d246bfdfcd96db"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:RQ2AC355JH2IR2HF7YOGOL6FCS","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Multi-turn Reinforcement Learning from Preference Human Feedback","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Asaf Cassel, Avinatan Hassidim, Avital Zipori, Aviv Rosenberg, Bilal Piot, Daniele Calandriello, Hila Noga, Idan Szpektor, Lior Shani, Oran Lang, Orgad Keller, R\\'emi Munos, Yossi Matias","submitted_at":"2024-05-23T14:53:54Z","abstract_excerpt":"Reinforcement Learning from Human Feedback (RLHF) has become the standard approach for aligning Large Language Models (LLMs) with human preferences, allowing LLMs to demonstrate remarkable abilities in various tasks. Existing methods work by emulating the preferences at the single decision (turn) level, limiting their capabilities in settings that require planning or multi-turn interactions to achieve a long-term goal. In this paper, we address this issue by developing novel methods for Reinforcement Learning (RL) from preference feedback between two full multi-turn conversations. In the tabul"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.14655","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.14655/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:43:07Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"tm4t+JmQc8rabVv4sVEDrIk+vg5rFdCbhCZKmh+5+0OaAvi+x93bV8ykpteYD8FAiy4D3d0rGRYs/C/piI3rBQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-08T08:49:34.568244Z"},"content_sha256":"7043f98cb5ce5f996909616652d12843d1ac3453c46012b8a9dba4c04c554551","schema_version":"1.0","event_id":"sha256:7043f98cb5ce5f996909616652d12843d1ac3453c46012b8a9dba4c04c554551"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/RQ2AC355JH2IR2HF7YOGOL6FCS/bundle.json","state_url":"https://pith.science/pith/RQ2AC355JH2IR2HF7YOGOL6FCS/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/RQ2AC355JH2IR2HF7YOGOL6FCS/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-08T08:49:34Z","links":{"resolver":"https://pith.science/pith/RQ2AC355JH2IR2HF7YOGOL6FCS","bundle":"https://pith.science/pith/RQ2AC355JH2IR2HF7YOGOL6FCS/bundle.json","state":"https://pith.science/pith/RQ2AC355JH2IR2HF7YOGOL6FCS/state.json","well_known_bundle":"https://pith.science/.well-known/pith/RQ2AC355JH2IR2HF7YOGOL6FCS/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:RQ2AC355JH2IR2HF7YOGOL6FCS","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"8abb2d05e5af6a010b8e7633cb06f477d5d33a8cd4891e51ba3d1b9d31d8aff3","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-05-23T14:53:54Z","title_canon_sha256":"2b85a3456dff451dd108feb3442e5f9f1567ff161086a03923b9afde920b263b"},"schema_version":"1.0","source":{"id":"2405.14655","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2405.14655","created_at":"2026-07-05T09:43:07Z"},{"alias_kind":"arxiv_version","alias_value":"2405.14655v2","created_at":"2026-07-05T09:43:07Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.14655","created_at":"2026-07-05T09:43:07Z"},{"alias_kind":"pith_short_12","alias_value":"RQ2AC355JH2I","created_at":"2026-07-05T09:43:07Z"},{"alias_kind":"pith_short_16","alias_value":"RQ2AC355JH2IR2HF","created_at":"2026-07-05T09:43:07Z"},{"alias_kind":"pith_short_8","alias_value":"RQ2AC355","created_at":"2026-07-05T09:43:07Z"}],"graph_snapshots":[{"event_id":"sha256:7043f98cb5ce5f996909616652d12843d1ac3453c46012b8a9dba4c04c554551","target":"graph","created_at":"2026-07-05T09:43:07Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2405.14655/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement Learning from Human Feedback (RLHF) has become the standard approach for aligning Large Language Models (LLMs) with human preferences, allowing LLMs to demonstrate remarkable abilities in various tasks. Existing methods work by emulating the preferences at the single decision (turn) level, limiting their capabilities in settings that require planning or multi-turn interactions to achieve a long-term goal. In this paper, we address this issue by developing novel methods for Reinforcement Learning (RL) from preference feedback between two full multi-turn conversations. In the tabul","authors_text":"Asaf Cassel, Avinatan Hassidim, Avital Zipori, Aviv Rosenberg, Bilal Piot, Daniele Calandriello, Hila Noga, Idan Szpektor, Lior Shani, Oran Lang, Orgad Keller, R\\'emi Munos, Yossi Matias","cross_cats":[],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-05-23T14:53:54Z","title":"Multi-turn Reinforcement Learning from Preference Human Feedback"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.14655","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:efa033d400ff99e49ac0e252946ef146b3497fdbac23e8cb76d246bfdfcd96db","target":"record","created_at":"2026-07-05T09:43:07Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"8abb2d05e5af6a010b8e7633cb06f477d5d33a8cd4891e51ba3d1b9d31d8aff3","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-05-23T14:53:54Z","title_canon_sha256":"2b85a3456dff451dd108feb3442e5f9f1567ff161086a03923b9afde920b263b"},"schema_version":"1.0","source":{"id":"2405.14655","kind":"arxiv","version":2}},"canonical_sha256":"8c34016fbd49f488e8e5fe1c672fc5148db70033f17fbe9b31bde71a9e217e60","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"8c34016fbd49f488e8e5fe1c672fc5148db70033f17fbe9b31bde71a9e217e60","first_computed_at":"2026-07-05T09:43:07.280550Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T09:43:07.280550Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"tfzBhaGVIV8HDAOeI8YcoZtv1Zi4vQPgJaPAVBuDvQQnb+Iu8gihL6Tp/9wA5h4xn8hx5cxhOd+/3+wH9cunBw==","signature_status":"signed_v1","signed_at":"2026-07-05T09:43:07.282422Z","signed_message":"canonical_sha256_bytes"},"source_id":"2405.14655","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:efa033d400ff99e49ac0e252946ef146b3497fdbac23e8cb76d246bfdfcd96db","sha256:7043f98cb5ce5f996909616652d12843d1ac3453c46012b8a9dba4c04c554551"],"state_sha256":"d60bbf04ce1a74bf644f694bb7830a144e565ff4ed198301c6b18838c31b1a32"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"9K3bd7Bh5L+fnz8CqFFESo/JELxcpLL8p81fWyQiK5KVloRJ918WnUQ3eR7talhRLSISXUV2rt9GXcB85YigCw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-08T08:49:34.577618Z","bundle_sha256":"fa5841fbc75b8db9c7e510e63291ea30654bb7b10036f7a859dd3338d7d341dc"}}