{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:DJDPY7MEZ3LZUNTJBSZVRIB4F7","short_pith_number":"pith:DJDPY7ME","schema_version":"1.0","canonical_sha256":"1a46fc7d84ced79a36690cb358a03c2ff6ced3bd9cdf23e4c9451a8d1169410c","source":{"kind":"arxiv","id":"2402.10571","version":2},"attestation_state":"computed","paper":{"title":"Direct Preference Optimization with an Offset","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Afra Amini, Ryan Cotterell, Tim Vieira","submitted_at":"2024-02-16T10:55:38Z","abstract_excerpt":"Direct preference optimization (DPO) is a successful fine-tuning strategy for aligning large language models with human preferences without the need to train a reward model or employ reinforcement learning. DPO, as originally formulated, relies on binary preference data and fine-tunes a language model to increase the likelihood of a preferred response over a dispreferred response. However, not all preference pairs are equal. Sometimes, the preferred response is only slightly better than the dispreferred one. In other cases, the preference is much stronger. For instance, if a response contains "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.10571","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-02-16T10:55:38Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"49354f850619f42eeb8bbed4042883dcaebf2bfae0da26d28f86cadea6917047","abstract_canon_sha256":"459612660cdfcc719a58946adedca45dde6b771c2e6a7d38b1dd25513cf5651f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:28:06.065231Z","signature_b64":"890PhpCgjs3ak7wjjKOYISlE5ed9OvmGny4dU/MXVK9INA3DJ2QM5CgJKgtbRdoHeUExrjBN71OwSuEM6ou1Dw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1a46fc7d84ced79a36690cb358a03c2ff6ced3bd9cdf23e4c9451a8d1169410c","last_reissued_at":"2026-07-05T08:28:06.064828Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:28:06.064828Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Direct Preference Optimization with an Offset","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Afra Amini, Ryan Cotterell, Tim Vieira","submitted_at":"2024-02-16T10:55:38Z","abstract_excerpt":"Direct preference optimization (DPO) is a successful fine-tuning strategy for aligning large language models with human preferences without the need to train a reward model or employ reinforcement learning. DPO, as originally formulated, relies on binary preference data and fine-tunes a language model to increase the likelihood of a preferred response over a dispreferred response. However, not all preference pairs are equal. Sometimes, the preferred response is only slightly better than the dispreferred one. In other cases, the preference is much stronger. For instance, if a response contains "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.10571","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.10571/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.10571","created_at":"2026-07-05T08:28:06.064883+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.10571v2","created_at":"2026-07-05T08:28:06.064883+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.10571","created_at":"2026-07-05T08:28:06.064883+00:00"},{"alias_kind":"pith_short_12","alias_value":"DJDPY7MEZ3LZ","created_at":"2026-07-05T08:28:06.064883+00:00"},{"alias_kind":"pith_short_16","alias_value":"DJDPY7MEZ3LZUNTJ","created_at":"2026-07-05T08:28:06.064883+00:00"},{"alias_kind":"pith_short_8","alias_value":"DJDPY7ME","created_at":"2026-07-05T08:28:06.064883+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2509.03526","citing_title":"Enhancing Speech Large Language Models through Reinforced Behavior Alignment","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2501.09686","citing_title":"Towards Large Reasoning Models: A Survey of Reinforced Reasoning with Large Language Models","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11974","citing_title":"Towards Order Fairness: Mitigating LLMs Order Sensitivity through Dual Group Advantage Optimization","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08472","citing_title":"Mid-Training with Self-Generated Data Improves Reinforcement Learning in Language Models","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10784","citing_title":"MASS-DPO: Multi-negative Active Sample Selection for Direct Policy Optimization","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02200","citing_title":"ARGUS: Policy-Adaptive Ad Governance via Evolving Reinforcement with Adversarial Umpiring","ref_index":72,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DJDPY7MEZ3LZUNTJBSZVRIB4F7","json":"https://pith.science/pith/DJDPY7MEZ3LZUNTJBSZVRIB4F7.json","graph_json":"https://pith.science/api/pith-number/DJDPY7MEZ3LZUNTJBSZVRIB4F7/graph.json","events_json":"https://pith.science/api/pith-number/DJDPY7MEZ3LZUNTJBSZVRIB4F7/events.json","paper":"https://pith.science/paper/DJDPY7ME"},"agent_actions":{"view_html":"https://pith.science/pith/DJDPY7MEZ3LZUNTJBSZVRIB4F7","download_json":"https://pith.science/pith/DJDPY7MEZ3LZUNTJBSZVRIB4F7.json","view_paper":"https://pith.science/paper/DJDPY7ME","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.10571&json=true","fetch_graph":"https://pith.science/api/pith-number/DJDPY7MEZ3LZUNTJBSZVRIB4F7/graph.json","fetch_events":"https://pith.science/api/pith-number/DJDPY7MEZ3LZUNTJBSZVRIB4F7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DJDPY7MEZ3LZUNTJBSZVRIB4F7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DJDPY7MEZ3LZUNTJBSZVRIB4F7/action/storage_attestation","attest_author":"https://pith.science/pith/DJDPY7MEZ3LZUNTJBSZVRIB4F7/action/author_attestation","sign_citation":"https://pith.science/pith/DJDPY7MEZ3LZUNTJBSZVRIB4F7/action/citation_signature","submit_replication":"https://pith.science/pith/DJDPY7MEZ3LZUNTJBSZVRIB4F7/action/replication_record"}},"created_at":"2026-07-05T08:28:06.064883+00:00","updated_at":"2026-07-05T08:28:06.064883+00:00"}