{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:K235WED2K2BIJYEDFWZE2CHJTA","short_pith_number":"pith:K235WED2","schema_version":"1.0","canonical_sha256":"56b7db107a568284e0832db24d08e99805f2b3ed33d58c5a2d176ac2ae1d5cdd","source":{"kind":"arxiv","id":"2409.06411","version":2},"attestation_state":"computed","paper":{"title":"Length Desensitization in Direct Preference Optimization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Chengcheng Han, Jingang Wang, Jun Xu, Rongxiang Weng, Wei Liu, Xuezhi Cao, Xunliang Cai, Yang Bai","submitted_at":"2024-09-10T10:49:38Z","abstract_excerpt":"Direct Preference Optimization (DPO) is widely utilized in the Reinforcement Learning from Human Feedback (RLHF) phase to align Large Language Models (LLMs) with human preferences, thereby enhancing both their harmlessness and efficacy. However, it has been observed that DPO tends to over-optimize for verbosity, which can detrimentally affect both performance and user experience. In this paper, we conduct an in-depth theoretical analysis of DPO's optimization objective and reveal a strong correlation between its implicit reward and data length. This correlation misguides the optimization direc"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.06411","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-09-10T10:49:38Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"2e69cb7969f28230c802a487487c0c281e82a8f55950b814dcf893efcb713b7c","abstract_canon_sha256":"84a526a0252f42fc4af7318266a79aa58c03caa4d01eb5ab5a6dce9a3c1a7a02"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:41:41.419976Z","signature_b64":"AMVi4L8Fj21CEXBQI0blC0/PVaBvYHG+UvO7NQQ0W1I+4mQj+yiFfa3NB4cLL4JCbwu0sryOl81Jf5h2th/eCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"56b7db107a568284e0832db24d08e99805f2b3ed33d58c5a2d176ac2ae1d5cdd","last_reissued_at":"2026-07-05T09:41:41.419500Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:41:41.419500Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Length Desensitization in Direct Preference Optimization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Chengcheng Han, Jingang Wang, Jun Xu, Rongxiang Weng, Wei Liu, Xuezhi Cao, Xunliang Cai, Yang Bai","submitted_at":"2024-09-10T10:49:38Z","abstract_excerpt":"Direct Preference Optimization (DPO) is widely utilized in the Reinforcement Learning from Human Feedback (RLHF) phase to align Large Language Models (LLMs) with human preferences, thereby enhancing both their harmlessness and efficacy. However, it has been observed that DPO tends to over-optimize for verbosity, which can detrimentally affect both performance and user experience. In this paper, we conduct an in-depth theoretical analysis of DPO's optimization objective and reveal a strong correlation between its implicit reward and data length. This correlation misguides the optimization direc"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.06411","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.06411/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.06411","created_at":"2026-07-05T09:41:41.419557+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.06411v2","created_at":"2026-07-05T09:41:41.419557+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.06411","created_at":"2026-07-05T09:41:41.419557+00:00"},{"alias_kind":"pith_short_12","alias_value":"K235WED2K2BI","created_at":"2026-07-05T09:41:41.419557+00:00"},{"alias_kind":"pith_short_16","alias_value":"K235WED2K2BIJYED","created_at":"2026-07-05T09:41:41.419557+00:00"},{"alias_kind":"pith_short_8","alias_value":"K235WED2","created_at":"2026-07-05T09:41:41.419557+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21943","citing_title":"Modularized Reinforcement Learning on LLMs: From MDP Creation to Exploration and Learning","ref_index":121,"is_internal_anchor":false},{"citing_arxiv_id":"2603.22267","citing_title":"TiCo: Time-Controllable Spoken Dialogue Model","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23194","citing_title":"From Coarse to Fine: Self-Adaptive Hierarchical Planning for LLM Agents","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15602","citing_title":"GroupDPO: Memory efficient Group-wise Direct Preference Optimization","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17197","citing_title":"Learning to Control Summaries with Score Ranking","ref_index":45,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/K235WED2K2BIJYEDFWZE2CHJTA","json":"https://pith.science/pith/K235WED2K2BIJYEDFWZE2CHJTA.json","graph_json":"https://pith.science/api/pith-number/K235WED2K2BIJYEDFWZE2CHJTA/graph.json","events_json":"https://pith.science/api/pith-number/K235WED2K2BIJYEDFWZE2CHJTA/events.json","paper":"https://pith.science/paper/K235WED2"},"agent_actions":{"view_html":"https://pith.science/pith/K235WED2K2BIJYEDFWZE2CHJTA","download_json":"https://pith.science/pith/K235WED2K2BIJYEDFWZE2CHJTA.json","view_paper":"https://pith.science/paper/K235WED2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.06411&json=true","fetch_graph":"https://pith.science/api/pith-number/K235WED2K2BIJYEDFWZE2CHJTA/graph.json","fetch_events":"https://pith.science/api/pith-number/K235WED2K2BIJYEDFWZE2CHJTA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/K235WED2K2BIJYEDFWZE2CHJTA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/K235WED2K2BIJYEDFWZE2CHJTA/action/storage_attestation","attest_author":"https://pith.science/pith/K235WED2K2BIJYEDFWZE2CHJTA/action/author_attestation","sign_citation":"https://pith.science/pith/K235WED2K2BIJYEDFWZE2CHJTA/action/citation_signature","submit_replication":"https://pith.science/pith/K235WED2K2BIJYEDFWZE2CHJTA/action/replication_record"}},"created_at":"2026-07-05T09:41:41.419557+00:00","updated_at":"2026-07-05T09:41:41.419557+00:00"}