{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:MNN2C64NX3ZVAUC3YMNHHCQEBM","short_pith_number":"pith:MNN2C64N","schema_version":"1.0","canonical_sha256":"635ba17b8dbef350505bc31a738a040b2eb53269c0a08972af54f9721d90d1c2","source":{"kind":"arxiv","id":"2406.09279","version":2},"attestation_state":"computed","paper":{"title":"Unpacking DPO and PPO: Disentangling Best Practices for Learning from Preference Feedback","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hamish Ivison, Hannaneh Hajishirzi, Jiacheng Liu, Nathan Lambert, Noah A. Smith, Valentina Pyatkin, Yejin Choi, Yizhong Wang, Zeqiu Wu","submitted_at":"2024-06-13T16:17:21Z","abstract_excerpt":"Learning from preference feedback has emerged as an essential step for improving the generation quality and performance of modern language models (LMs). Despite its widespread use, the way preference-based learning is applied varies wildly, with differing data, learning algorithms, and evaluations used, making disentangling the impact of each aspect difficult. In this work, we identify four core aspects of preference-based learning: preference data, learning algorithm, reward model, and policy training prompts, systematically investigate the impact of these components on downstream model perfo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.09279","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-13T16:17:21Z","cross_cats_sorted":[],"title_canon_sha256":"1cd3bfc79394695e3b7c836bf9648ef0bfd3715b4317f362c3f863db7f60a21e","abstract_canon_sha256":"fb57553bdf4dd687e6a7cfbddc2163e4f4b8a7d5751bc72c3d97c716b2928678"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:17:13.878882Z","signature_b64":"8++xmidi/3+ydkg/jezJ6FjjjysW2Diguj4zSHbjsYYHR/Kys69ykHjqPZ8K6NA1ZauwRFMIhMOZOhob8s2BCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"635ba17b8dbef350505bc31a738a040b2eb53269c0a08972af54f9721d90d1c2","last_reissued_at":"2026-07-05T09:17:13.878437Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:17:13.878437Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Unpacking DPO and PPO: Disentangling Best Practices for Learning from Preference Feedback","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hamish Ivison, Hannaneh Hajishirzi, Jiacheng Liu, Nathan Lambert, Noah A. Smith, Valentina Pyatkin, Yejin Choi, Yizhong Wang, Zeqiu Wu","submitted_at":"2024-06-13T16:17:21Z","abstract_excerpt":"Learning from preference feedback has emerged as an essential step for improving the generation quality and performance of modern language models (LMs). Despite its widespread use, the way preference-based learning is applied varies wildly, with differing data, learning algorithms, and evaluations used, making disentangling the impact of each aspect difficult. In this work, we identify four core aspects of preference-based learning: preference data, learning algorithm, reward model, and policy training prompts, systematically investigate the impact of these components on downstream model perfo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.09279","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.09279/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.09279","created_at":"2026-07-05T09:17:13.878492+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.09279v2","created_at":"2026-07-05T09:17:13.878492+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.09279","created_at":"2026-07-05T09:17:13.878492+00:00"},{"alias_kind":"pith_short_12","alias_value":"MNN2C64NX3ZV","created_at":"2026-07-05T09:17:13.878492+00:00"},{"alias_kind":"pith_short_16","alias_value":"MNN2C64NX3ZVAUC3","created_at":"2026-07-05T09:17:13.878492+00:00"},{"alias_kind":"pith_short_8","alias_value":"MNN2C64N","created_at":"2026-07-05T09:17:13.878492+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2502.02737","citing_title":"SmolLM2: When Smol Goes Big -- Data-Centric Training of a Small Language Model","ref_index":182,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00195","citing_title":"Diversity in Large Language Models under Supervised Fine-Tuning","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00195","citing_title":"Diversity in Large Language Models under Supervised Fine-Tuning","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04291","citing_title":"Leveraging Pretrained Language Models as Energy Functions for Glauber Dynamics Text Diffusion","ref_index":50,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MNN2C64NX3ZVAUC3YMNHHCQEBM","json":"https://pith.science/pith/MNN2C64NX3ZVAUC3YMNHHCQEBM.json","graph_json":"https://pith.science/api/pith-number/MNN2C64NX3ZVAUC3YMNHHCQEBM/graph.json","events_json":"https://pith.science/api/pith-number/MNN2C64NX3ZVAUC3YMNHHCQEBM/events.json","paper":"https://pith.science/paper/MNN2C64N"},"agent_actions":{"view_html":"https://pith.science/pith/MNN2C64NX3ZVAUC3YMNHHCQEBM","download_json":"https://pith.science/pith/MNN2C64NX3ZVAUC3YMNHHCQEBM.json","view_paper":"https://pith.science/paper/MNN2C64N","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.09279&json=true","fetch_graph":"https://pith.science/api/pith-number/MNN2C64NX3ZVAUC3YMNHHCQEBM/graph.json","fetch_events":"https://pith.science/api/pith-number/MNN2C64NX3ZVAUC3YMNHHCQEBM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MNN2C64NX3ZVAUC3YMNHHCQEBM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MNN2C64NX3ZVAUC3YMNHHCQEBM/action/storage_attestation","attest_author":"https://pith.science/pith/MNN2C64NX3ZVAUC3YMNHHCQEBM/action/author_attestation","sign_citation":"https://pith.science/pith/MNN2C64NX3ZVAUC3YMNHHCQEBM/action/citation_signature","submit_replication":"https://pith.science/pith/MNN2C64NX3ZVAUC3YMNHHCQEBM/action/replication_record"}},"created_at":"2026-07-05T09:17:13.878492+00:00","updated_at":"2026-07-05T09:17:13.878492+00:00"}