{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:JUIHIIJNNE5BPADSDCSVBOQYE2","short_pith_number":"pith:JUIHIIJN","canonical_record":{"source":{"id":"2409.03650","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-09-05T16:08:19Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"e9aa2aaf5227fb95d6d872bbea5af176e8411fe5686d05e65bcbb3edd9982f88","abstract_canon_sha256":"dfdfad3b1ad8e96773c39abc1a8f788fa07a7105d29ee29b3c87243cc369491f"},"schema_version":"1.0"},"canonical_sha256":"4d1074212d693a17807218a550ba1826971ba5362ab11e96df3eb9fba46b7b9a","source":{"kind":"arxiv","id":"2409.03650","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2409.03650","created_at":"2026-07-05T09:15:14Z"},{"alias_kind":"arxiv_version","alias_value":"2409.03650v2","created_at":"2026-07-05T09:15:14Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.03650","created_at":"2026-07-05T09:15:14Z"},{"alias_kind":"pith_short_12","alias_value":"JUIHIIJNNE5B","created_at":"2026-07-05T09:15:14Z"},{"alias_kind":"pith_short_16","alias_value":"JUIHIIJNNE5BPADS","created_at":"2026-07-05T09:15:14Z"},{"alias_kind":"pith_short_8","alias_value":"JUIHIIJN","created_at":"2026-07-05T09:15:14Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:JUIHIIJNNE5BPADSDCSVBOQYE2","target":"record","payload":{"canonical_record":{"source":{"id":"2409.03650","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-09-05T16:08:19Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"e9aa2aaf5227fb95d6d872bbea5af176e8411fe5686d05e65bcbb3edd9982f88","abstract_canon_sha256":"dfdfad3b1ad8e96773c39abc1a8f788fa07a7105d29ee29b3c87243cc369491f"},"schema_version":"1.0"},"canonical_sha256":"4d1074212d693a17807218a550ba1826971ba5362ab11e96df3eb9fba46b7b9a","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:15:14.348704Z","signature_b64":"BFTEikPrX7TH8+/If92gGwbVDu5ZDY9np5vgnwXpAFG7ZNWOqgjLN30LmWInWpCFP75vShr76eAPbW/ZLstOCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4d1074212d693a17807218a550ba1826971ba5362ab11e96df3eb9fba46b7b9a","last_reissued_at":"2026-07-05T09:15:14.348217Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:15:14.348217Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2409.03650","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:15:14Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"IAQu2DEdbLXuGUrySdjGh057RP/rhbEzB1j78/KyEsTKtkPF6TF9CsswyQreWYDhFWi+9l7ZC4Vjd8AJIosfBQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T19:36:27.882289Z"},"content_sha256":"47cb9370361c2bf58b2df1270ed75b3c829d04d863c5ffe3cf28bf7cf0751803","schema_version":"1.0","event_id":"sha256:47cb9370361c2bf58b2df1270ed75b3c829d04d863c5ffe3cf28bf7cf0751803"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:JUIHIIJNNE5BPADSDCSVBOQYE2","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"On the Limited Generalization Capability of the Implicit Reward Model Induced by Direct Preference Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Barry-John Theobald, Chen Huang, Katherine Metcalf, Maartje ter Hoeve, Skyler Seto, Tong Zhang, Xuan Wang, Yizhe Zhang, Yong Lin","submitted_at":"2024-09-05T16:08:19Z","abstract_excerpt":"Reinforcement Learning from Human Feedback (RLHF) is an effective approach for aligning language models to human preferences. Central to RLHF is learning a reward function for scoring human preferences. Two main approaches for learning a reward model are 1) training an EXplicit Reward Model (EXRM) as in RLHF, and 2) using an implicit reward learned from preference data through methods such as Direct Preference Optimization (DPO). Prior work has shown that the implicit reward model of DPO (denoted as DPORM) can approximate an EXRM in the limit. DPORM's effectiveness directly implies the optimal"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.03650","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.03650/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:15:14Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"iAP9Z4ygkWwfFbwerNa8b5uTc4aujzdJ3/vG01pHG8Gjks3PxRHPZ/biWUoxgT5yVsjuM8Z3bvtiViTWfCRRDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T19:36:27.883170Z"},"content_sha256":"fdf0dcd83a5fc780eedc2ac46495a680f480992ddd366a5effdb3dd44920b74a","schema_version":"1.0","event_id":"sha256:fdf0dcd83a5fc780eedc2ac46495a680f480992ddd366a5effdb3dd44920b74a"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/JUIHIIJNNE5BPADSDCSVBOQYE2/bundle.json","state_url":"https://pith.science/pith/JUIHIIJNNE5BPADSDCSVBOQYE2/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/JUIHIIJNNE5BPADSDCSVBOQYE2/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-07T19:36:27Z","links":{"resolver":"https://pith.science/pith/JUIHIIJNNE5BPADSDCSVBOQYE2","bundle":"https://pith.science/pith/JUIHIIJNNE5BPADSDCSVBOQYE2/bundle.json","state":"https://pith.science/pith/JUIHIIJNNE5BPADSDCSVBOQYE2/state.json","well_known_bundle":"https://pith.science/.well-known/pith/JUIHIIJNNE5BPADSDCSVBOQYE2/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:JUIHIIJNNE5BPADSDCSVBOQYE2","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"dfdfad3b1ad8e96773c39abc1a8f788fa07a7105d29ee29b3c87243cc369491f","cross_cats_sorted":["cs.CL"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-09-05T16:08:19Z","title_canon_sha256":"e9aa2aaf5227fb95d6d872bbea5af176e8411fe5686d05e65bcbb3edd9982f88"},"schema_version":"1.0","source":{"id":"2409.03650","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2409.03650","created_at":"2026-07-05T09:15:14Z"},{"alias_kind":"arxiv_version","alias_value":"2409.03650v2","created_at":"2026-07-05T09:15:14Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.03650","created_at":"2026-07-05T09:15:14Z"},{"alias_kind":"pith_short_12","alias_value":"JUIHIIJNNE5B","created_at":"2026-07-05T09:15:14Z"},{"alias_kind":"pith_short_16","alias_value":"JUIHIIJNNE5BPADS","created_at":"2026-07-05T09:15:14Z"},{"alias_kind":"pith_short_8","alias_value":"JUIHIIJN","created_at":"2026-07-05T09:15:14Z"}],"graph_snapshots":[{"event_id":"sha256:fdf0dcd83a5fc780eedc2ac46495a680f480992ddd366a5effdb3dd44920b74a","target":"graph","created_at":"2026-07-05T09:15:14Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2409.03650/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement Learning from Human Feedback (RLHF) is an effective approach for aligning language models to human preferences. Central to RLHF is learning a reward function for scoring human preferences. Two main approaches for learning a reward model are 1) training an EXplicit Reward Model (EXRM) as in RLHF, and 2) using an implicit reward learned from preference data through methods such as Direct Preference Optimization (DPO). Prior work has shown that the implicit reward model of DPO (denoted as DPORM) can approximate an EXRM in the limit. DPORM's effectiveness directly implies the optimal","authors_text":"Barry-John Theobald, Chen Huang, Katherine Metcalf, Maartje ter Hoeve, Skyler Seto, Tong Zhang, Xuan Wang, Yizhe Zhang, Yong Lin","cross_cats":["cs.CL"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-09-05T16:08:19Z","title":"On the Limited Generalization Capability of the Implicit Reward Model Induced by Direct Preference Optimization"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.03650","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:47cb9370361c2bf58b2df1270ed75b3c829d04d863c5ffe3cf28bf7cf0751803","target":"record","created_at":"2026-07-05T09:15:14Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"dfdfad3b1ad8e96773c39abc1a8f788fa07a7105d29ee29b3c87243cc369491f","cross_cats_sorted":["cs.CL"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-09-05T16:08:19Z","title_canon_sha256":"e9aa2aaf5227fb95d6d872bbea5af176e8411fe5686d05e65bcbb3edd9982f88"},"schema_version":"1.0","source":{"id":"2409.03650","kind":"arxiv","version":2}},"canonical_sha256":"4d1074212d693a17807218a550ba1826971ba5362ab11e96df3eb9fba46b7b9a","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"4d1074212d693a17807218a550ba1826971ba5362ab11e96df3eb9fba46b7b9a","first_computed_at":"2026-07-05T09:15:14.348217Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T09:15:14.348217Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"BFTEikPrX7TH8+/If92gGwbVDu5ZDY9np5vgnwXpAFG7ZNWOqgjLN30LmWInWpCFP75vShr76eAPbW/ZLstOCQ==","signature_status":"signed_v1","signed_at":"2026-07-05T09:15:14.348704Z","signed_message":"canonical_sha256_bytes"},"source_id":"2409.03650","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:47cb9370361c2bf58b2df1270ed75b3c829d04d863c5ffe3cf28bf7cf0751803","sha256:fdf0dcd83a5fc780eedc2ac46495a680f480992ddd366a5effdb3dd44920b74a"],"state_sha256":"ab6a4080ee6c45668469a4cbf6d5ec2a68e191e89f1c437a8fbeb8538c8eb390"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"svPLKAV9T6i2Ov1EZIIJTiEK2T8v/zR3GtYB7SBhsikeszfRlm5Hi2GLjHHmCtdLtoXd2daXvA0qde9m0nq5Cw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-07T19:36:27.890049Z","bundle_sha256":"33230a2c88feba312baac8e54f6a991044eacf273f9257ed07028e7a128cc43c"}}