{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:JUIHIIJNNE5BPADSDCSVBOQYE2","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"dfdfad3b1ad8e96773c39abc1a8f788fa07a7105d29ee29b3c87243cc369491f","cross_cats_sorted":["cs.CL"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-09-05T16:08:19Z","title_canon_sha256":"e9aa2aaf5227fb95d6d872bbea5af176e8411fe5686d05e65bcbb3edd9982f88"},"schema_version":"1.0","source":{"id":"2409.03650","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2409.03650","created_at":"2026-07-05T09:15:14Z"},{"alias_kind":"arxiv_version","alias_value":"2409.03650v2","created_at":"2026-07-05T09:15:14Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.03650","created_at":"2026-07-05T09:15:14Z"},{"alias_kind":"pith_short_12","alias_value":"JUIHIIJNNE5B","created_at":"2026-07-05T09:15:14Z"},{"alias_kind":"pith_short_16","alias_value":"JUIHIIJNNE5BPADS","created_at":"2026-07-05T09:15:14Z"},{"alias_kind":"pith_short_8","alias_value":"JUIHIIJN","created_at":"2026-07-05T09:15:14Z"}],"graph_snapshots":[{"event_id":"sha256:fdf0dcd83a5fc780eedc2ac46495a680f480992ddd366a5effdb3dd44920b74a","target":"graph","created_at":"2026-07-05T09:15:14Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2409.03650/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement Learning from Human Feedback (RLHF) is an effective approach for aligning language models to human preferences. Central to RLHF is learning a reward function for scoring human preferences. Two main approaches for learning a reward model are 1) training an EXplicit Reward Model (EXRM) as in RLHF, and 2) using an implicit reward learned from preference data through methods such as Direct Preference Optimization (DPO). Prior work has shown that the implicit reward model of DPO (denoted as DPORM) can approximate an EXRM in the limit. DPORM's effectiveness directly implies the optimal","authors_text":"Barry-John Theobald, Chen Huang, Katherine Metcalf, Maartje ter Hoeve, Skyler Seto, Tong Zhang, Xuan Wang, Yizhe Zhang, Yong Lin","cross_cats":["cs.CL"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-09-05T16:08:19Z","title":"On the Limited Generalization Capability of the Implicit Reward Model Induced by Direct Preference Optimization"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.03650","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:47cb9370361c2bf58b2df1270ed75b3c829d04d863c5ffe3cf28bf7cf0751803","target":"record","created_at":"2026-07-05T09:15:14Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"dfdfad3b1ad8e96773c39abc1a8f788fa07a7105d29ee29b3c87243cc369491f","cross_cats_sorted":["cs.CL"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-09-05T16:08:19Z","title_canon_sha256":"e9aa2aaf5227fb95d6d872bbea5af176e8411fe5686d05e65bcbb3edd9982f88"},"schema_version":"1.0","source":{"id":"2409.03650","kind":"arxiv","version":2}},"canonical_sha256":"4d1074212d693a17807218a550ba1826971ba5362ab11e96df3eb9fba46b7b9a","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"4d1074212d693a17807218a550ba1826971ba5362ab11e96df3eb9fba46b7b9a","first_computed_at":"2026-07-05T09:15:14.348217Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T09:15:14.348217Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"BFTEikPrX7TH8+/If92gGwbVDu5ZDY9np5vgnwXpAFG7ZNWOqgjLN30LmWInWpCFP75vShr76eAPbW/ZLstOCQ==","signature_status":"signed_v1","signed_at":"2026-07-05T09:15:14.348704Z","signed_message":"canonical_sha256_bytes"},"source_id":"2409.03650","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:47cb9370361c2bf58b2df1270ed75b3c829d04d863c5ffe3cf28bf7cf0751803","sha256:fdf0dcd83a5fc780eedc2ac46495a680f480992ddd366a5effdb3dd44920b74a"],"state_sha256":"ab6a4080ee6c45668469a4cbf6d5ec2a68e191e89f1c437a8fbeb8538c8eb390"}