{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:T3CPF6FMDSCLLXOAY3EJM76TLK","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"a0cf55c17006fac0f8939ca45261d7d5014dcccee3acd1f11273dfe336b6d5ff","cross_cats_sorted":["stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-02-11T21:44:21Z","title_canon_sha256":"6bc8a4e8b396505d047129dc2602cc5f31387418a2e15d2cf7ed85386fbd48e7"},"schema_version":"1.0","source":{"id":"2402.07314","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2402.07314","created_at":"2026-07-05T09:34:11Z"},{"alias_kind":"arxiv_version","alias_value":"2402.07314v3","created_at":"2026-07-05T09:34:11Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.07314","created_at":"2026-07-05T09:34:11Z"},{"alias_kind":"pith_short_12","alias_value":"T3CPF6FMDSCL","created_at":"2026-07-05T09:34:11Z"},{"alias_kind":"pith_short_16","alias_value":"T3CPF6FMDSCLLXOA","created_at":"2026-07-05T09:34:11Z"},{"alias_kind":"pith_short_8","alias_value":"T3CPF6FM","created_at":"2026-07-05T09:34:11Z"}],"graph_snapshots":[{"event_id":"sha256:0c4562fb425f7786bfee5ecbd757683876bdda85286e396892f8a591b3174650","target":"graph","created_at":"2026-07-05T09:34:11Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2402.07314/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"We investigate Reinforcement Learning from Human Feedback (RLHF) in the context of a general preference oracle. In particular, we do not assume the existence of a reward function and an oracle preference signal drawn from the Bradley-Terry model as most of the prior works do. We consider a standard mathematical formulation, the reverse-KL regularized minimax game between two LLMs for RLHF under general preference oracle. The learning objective of this formulation is to find a policy so that it is consistently preferred by the KL-regularized preference oracle over any competing LLMs. We show th","authors_text":"Chenlu Ye, HanZe Dong, Nan Jiang, Tong Zhang, Wei Xiong, Yuheng Zhang","cross_cats":["stat.ML"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-02-11T21:44:21Z","title":"Online Iterative Reinforcement Learning from Human Feedback with General Preference Model"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.07314","kind":"arxiv","version":3},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:913a5e3721d79f0f23b20dfa6251c05c3ceaf62541a734c7469aed89bfa0ddd1","target":"record","created_at":"2026-07-05T09:34:11Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"a0cf55c17006fac0f8939ca45261d7d5014dcccee3acd1f11273dfe336b6d5ff","cross_cats_sorted":["stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-02-11T21:44:21Z","title_canon_sha256":"6bc8a4e8b396505d047129dc2602cc5f31387418a2e15d2cf7ed85386fbd48e7"},"schema_version":"1.0","source":{"id":"2402.07314","kind":"arxiv","version":3}},"canonical_sha256":"9ec4f2f8ac1c84b5ddc0c6c8967fd35ab83ba8042c2c814224a228acb1b01862","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"9ec4f2f8ac1c84b5ddc0c6c8967fd35ab83ba8042c2c814224a228acb1b01862","first_computed_at":"2026-07-05T09:34:11.422939Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T09:34:11.422939Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"cdWFKyGBgR5/ot7IqWRaG3bIdhuvga/sZYS/wr9SlOaipLd6Qu3SJlhMQd/mnabbkSd1PmJ6pivd+aXXD4YqAg==","signature_status":"signed_v1","signed_at":"2026-07-05T09:34:11.423468Z","signed_message":"canonical_sha256_bytes"},"source_id":"2402.07314","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:913a5e3721d79f0f23b20dfa6251c05c3ceaf62541a734c7469aed89bfa0ddd1","sha256:0c4562fb425f7786bfee5ecbd757683876bdda85286e396892f8a591b3174650"],"state_sha256":"9b3006480d12e9ffbac142d3dfb5a9bcf76859cba22d0d3864859e31bcc5b5eb"}