{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:4RWFQYLNVGBMNJGMFUI43ZMISO","short_pith_number":"pith:4RWFQYLN","canonical_record":{"source":{"id":"2405.19316","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-29T17:39:48Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"3832215bb66ec71f6c9a21a91b62a85c944418c4410e1bd4ea63c7354f79e83c","abstract_canon_sha256":"d518a85ee8c71d5e1bafaf81c45bde5b0667bf3908f88c547d45da2cdc4604a1"},"schema_version":"1.0"},"canonical_sha256":"e46c58616da982c6a4cc2d11cde58893a1d3c5aed8be6449540fea4262d20460","source":{"kind":"arxiv","id":"2405.19316","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2405.19316","created_at":"2026-07-05T10:22:46Z"},{"alias_kind":"arxiv_version","alias_value":"2405.19316v2","created_at":"2026-07-05T10:22:46Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.19316","created_at":"2026-07-05T10:22:46Z"},{"alias_kind":"pith_short_12","alias_value":"4RWFQYLNVGBM","created_at":"2026-07-05T10:22:46Z"},{"alias_kind":"pith_short_16","alias_value":"4RWFQYLNVGBMNJGM","created_at":"2026-07-05T10:22:46Z"},{"alias_kind":"pith_short_8","alias_value":"4RWFQYLN","created_at":"2026-07-05T10:22:46Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:4RWFQYLNVGBMNJGMFUI43ZMISO","target":"record","payload":{"canonical_record":{"source":{"id":"2405.19316","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-29T17:39:48Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"3832215bb66ec71f6c9a21a91b62a85c944418c4410e1bd4ea63c7354f79e83c","abstract_canon_sha256":"d518a85ee8c71d5e1bafaf81c45bde5b0667bf3908f88c547d45da2cdc4604a1"},"schema_version":"1.0"},"canonical_sha256":"e46c58616da982c6a4cc2d11cde58893a1d3c5aed8be6449540fea4262d20460","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:22:46.843367Z","signature_b64":"N9Fh8AvGkxHlLLf+1FJITnBe8HPQFsPaE8cwFFefgrZkW0xfVoKA3eoxpPvWa8mAYJc9TMrWeu5+HmTDBDqWCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e46c58616da982c6a4cc2d11cde58893a1d3c5aed8be6449540fea4262d20460","last_reissued_at":"2026-07-05T10:22:46.841026Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:22:46.841026Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2405.19316","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:22:46Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"jFAxJkA/uv7+lOSXu5Jk6Wpl6JYem0Xa6lnN6bgE8j1EH/LGqGhgDMbvyNSCLC5RIVoaLZvU8VIb3wqVvWE9CA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-08T23:11:22.986910Z"},"content_sha256":"32245fe1d67dced3fa52ab9bec3c3e961a668e170f0f4e5d1feb20f56a934201","schema_version":"1.0","event_id":"sha256:32245fe1d67dced3fa52ab9bec3c3e961a668e170f0f4e5d1feb20f56a934201"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:4RWFQYLNVGBMNJGMFUI43ZMISO","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Robust Preference Optimization through Reward Model Distillation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Adam Fisch, Ahmad Beirami, Alekh Agarwal, Chirag Nagpal, Jacob Eisenstein, Jonathan Berant, Pete Shaw, Vicky Zayats","submitted_at":"2024-05-29T17:39:48Z","abstract_excerpt":"Language model (LM) post-training (or alignment) involves maximizing a reward function that is derived from preference annotations. Direct Preference Optimization (DPO) is a popular offline alignment method that trains a policy directly on preference data without the need to train a reward model or apply reinforcement learning. However, the empirical evidence suggests that DPO typically assigns implicit rewards that overfit, and trend towards infinite magnitude. This frequently leads to degenerate policies, sometimes causing even the probabilities of the preferred generations to go to zero. In"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.19316","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.19316/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:22:46Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"IuJQPvdwywkGWfPMivGHhVZDfca/Bf5r+PMaCt+ugiPvuQElyVcSlQpuRUFJWM+zvLDD5aNEywBgzjP2QJX9AQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-08T23:11:22.987408Z"},"content_sha256":"9620255a71849cc03c49ba5cf53b7c464fff3958ea9da9c04bd86e20e2739dad","schema_version":"1.0","event_id":"sha256:9620255a71849cc03c49ba5cf53b7c464fff3958ea9da9c04bd86e20e2739dad"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/4RWFQYLNVGBMNJGMFUI43ZMISO/bundle.json","state_url":"https://pith.science/pith/4RWFQYLNVGBMNJGMFUI43ZMISO/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/4RWFQYLNVGBMNJGMFUI43ZMISO/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-08T23:11:22Z","links":{"resolver":"https://pith.science/pith/4RWFQYLNVGBMNJGMFUI43ZMISO","bundle":"https://pith.science/pith/4RWFQYLNVGBMNJGMFUI43ZMISO/bundle.json","state":"https://pith.science/pith/4RWFQYLNVGBMNJGMFUI43ZMISO/state.json","well_known_bundle":"https://pith.science/.well-known/pith/4RWFQYLNVGBMNJGMFUI43ZMISO/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:4RWFQYLNVGBMNJGMFUI43ZMISO","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"d518a85ee8c71d5e1bafaf81c45bde5b0667bf3908f88c547d45da2cdc4604a1","cross_cats_sorted":["cs.CL"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-29T17:39:48Z","title_canon_sha256":"3832215bb66ec71f6c9a21a91b62a85c944418c4410e1bd4ea63c7354f79e83c"},"schema_version":"1.0","source":{"id":"2405.19316","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2405.19316","created_at":"2026-07-05T10:22:46Z"},{"alias_kind":"arxiv_version","alias_value":"2405.19316v2","created_at":"2026-07-05T10:22:46Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.19316","created_at":"2026-07-05T10:22:46Z"},{"alias_kind":"pith_short_12","alias_value":"4RWFQYLNVGBM","created_at":"2026-07-05T10:22:46Z"},{"alias_kind":"pith_short_16","alias_value":"4RWFQYLNVGBMNJGM","created_at":"2026-07-05T10:22:46Z"},{"alias_kind":"pith_short_8","alias_value":"4RWFQYLN","created_at":"2026-07-05T10:22:46Z"}],"graph_snapshots":[{"event_id":"sha256:9620255a71849cc03c49ba5cf53b7c464fff3958ea9da9c04bd86e20e2739dad","target":"graph","created_at":"2026-07-05T10:22:46Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2405.19316/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Language model (LM) post-training (or alignment) involves maximizing a reward function that is derived from preference annotations. Direct Preference Optimization (DPO) is a popular offline alignment method that trains a policy directly on preference data without the need to train a reward model or apply reinforcement learning. However, the empirical evidence suggests that DPO typically assigns implicit rewards that overfit, and trend towards infinite magnitude. This frequently leads to degenerate policies, sometimes causing even the probabilities of the preferred generations to go to zero. In","authors_text":"Adam Fisch, Ahmad Beirami, Alekh Agarwal, Chirag Nagpal, Jacob Eisenstein, Jonathan Berant, Pete Shaw, Vicky Zayats","cross_cats":["cs.CL"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-29T17:39:48Z","title":"Robust Preference Optimization through Reward Model Distillation"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.19316","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:32245fe1d67dced3fa52ab9bec3c3e961a668e170f0f4e5d1feb20f56a934201","target":"record","created_at":"2026-07-05T10:22:46Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"d518a85ee8c71d5e1bafaf81c45bde5b0667bf3908f88c547d45da2cdc4604a1","cross_cats_sorted":["cs.CL"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-29T17:39:48Z","title_canon_sha256":"3832215bb66ec71f6c9a21a91b62a85c944418c4410e1bd4ea63c7354f79e83c"},"schema_version":"1.0","source":{"id":"2405.19316","kind":"arxiv","version":2}},"canonical_sha256":"e46c58616da982c6a4cc2d11cde58893a1d3c5aed8be6449540fea4262d20460","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"e46c58616da982c6a4cc2d11cde58893a1d3c5aed8be6449540fea4262d20460","first_computed_at":"2026-07-05T10:22:46.841026Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T10:22:46.841026Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"N9Fh8AvGkxHlLLf+1FJITnBe8HPQFsPaE8cwFFefgrZkW0xfVoKA3eoxpPvWa8mAYJc9TMrWeu5+HmTDBDqWCw==","signature_status":"signed_v1","signed_at":"2026-07-05T10:22:46.843367Z","signed_message":"canonical_sha256_bytes"},"source_id":"2405.19316","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:32245fe1d67dced3fa52ab9bec3c3e961a668e170f0f4e5d1feb20f56a934201","sha256:9620255a71849cc03c49ba5cf53b7c464fff3958ea9da9c04bd86e20e2739dad"],"state_sha256":"7f73aff9a25409737762caa973fb34752a740583c2232a2a6b79d9416e922f9a"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"6Laln3nxDI1ZgkCeYGzvA37XERE5qO+nS9v0BL1V/kRL03rjj37T8nHDsm549sppABz3Pfyu6kbUK9dE49nrBA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-08T23:11:22.991206Z","bundle_sha256":"5803b0cf8f2cffe7abd73767e5e8e17ba071aa5316aada3407daf0bef317e910"}}