{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:TVQ6NI7HK5VX53IWS7ER7LUEDO","short_pith_number":"pith:TVQ6NI7H","schema_version":"1.0","canonical_sha256":"9d61e6a3e7576b7eed1697c91fae841bb7d7c24c32e3d923c735375ba46b91fd","source":{"kind":"arxiv","id":"2404.04626","version":1},"attestation_state":"computed","paper":{"title":"Towards Analyzing and Understanding the Limitations of DPO: A Theoretical Perspective","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Bowen Qin, Chen Huang, Duanyu Feng, Wenqiang Lei, Zheng Zhang","submitted_at":"2024-04-06T13:24:37Z","abstract_excerpt":"Direct Preference Optimization (DPO), which derives reward signals directly from pairwise preference data, has shown its effectiveness on aligning Large Language Models (LLMs) with human preferences. Despite its widespread use across various tasks, DPO has been criticized for its sensitivity to the SFT's effectiveness and its hindrance to the learning capacity towards human-preferred responses, leading to less satisfactory performance. To overcome those limitations, the theoretical understanding of DPO are indispensable but still lacking. To this end, we take a step towards theoretically analy"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.04626","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-04-06T13:24:37Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"111eb93ae73dd3c1977fce5fc8a4ec2157ae3f4eed30156a30f2f4cd3fdfbbb5","abstract_canon_sha256":"aa566c110326665fbf1fb4aac9b6bada89c16cc412e5f163aa36a2881822bcd7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:05:19.293827Z","signature_b64":"Ed+m+Rb35IZXNPqvX6VWFhXVYYMpaDbngBHDhfTmIyJ4S3K6lUVcYaMZWp3Oxv/tW9bs4Sspc2aEFmU/oBDVAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9d61e6a3e7576b7eed1697c91fae841bb7d7c24c32e3d923c735375ba46b91fd","last_reissued_at":"2026-07-05T08:05:19.293479Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:05:19.293479Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards Analyzing and Understanding the Limitations of DPO: A Theoretical Perspective","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Bowen Qin, Chen Huang, Duanyu Feng, Wenqiang Lei, Zheng Zhang","submitted_at":"2024-04-06T13:24:37Z","abstract_excerpt":"Direct Preference Optimization (DPO), which derives reward signals directly from pairwise preference data, has shown its effectiveness on aligning Large Language Models (LLMs) with human preferences. Despite its widespread use across various tasks, DPO has been criticized for its sensitivity to the SFT's effectiveness and its hindrance to the learning capacity towards human-preferred responses, leading to less satisfactory performance. To overcome those limitations, the theoretical understanding of DPO are indispensable but still lacking. To this end, we take a step towards theoretically analy"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.04626","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.04626/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.04626","created_at":"2026-07-05T08:05:19.293533+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.04626v1","created_at":"2026-07-05T08:05:19.293533+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.04626","created_at":"2026-07-05T08:05:19.293533+00:00"},{"alias_kind":"pith_short_12","alias_value":"TVQ6NI7HK5VX","created_at":"2026-07-05T08:05:19.293533+00:00"},{"alias_kind":"pith_short_16","alias_value":"TVQ6NI7HK5VX53IW","created_at":"2026-07-05T08:05:19.293533+00:00"},{"alias_kind":"pith_short_8","alias_value":"TVQ6NI7H","created_at":"2026-07-05T08:05:19.293533+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.26315","citing_title":"Curriculum Learning for Safety Alignment","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28440","citing_title":"AdaDPO: Self-Adaptive Direct Preference Optimization with Balanced Gradient Updates","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2402.13228","citing_title":"Smaug: Fixing Failure Modes of Preference Optimisation with DPO-Positive","ref_index":117,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15012","citing_title":"Boosting Reinforcement Learning with Verifiable Rewards via Randomly Selected Few-Shot Guidance","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20933","citing_title":"IRIS: Interpolative R\\'enyi Iterative Self-play for Large Language Model Fine-Tuning","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TVQ6NI7HK5VX53IWS7ER7LUEDO","json":"https://pith.science/pith/TVQ6NI7HK5VX53IWS7ER7LUEDO.json","graph_json":"https://pith.science/api/pith-number/TVQ6NI7HK5VX53IWS7ER7LUEDO/graph.json","events_json":"https://pith.science/api/pith-number/TVQ6NI7HK5VX53IWS7ER7LUEDO/events.json","paper":"https://pith.science/paper/TVQ6NI7H"},"agent_actions":{"view_html":"https://pith.science/pith/TVQ6NI7HK5VX53IWS7ER7LUEDO","download_json":"https://pith.science/pith/TVQ6NI7HK5VX53IWS7ER7LUEDO.json","view_paper":"https://pith.science/paper/TVQ6NI7H","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.04626&json=true","fetch_graph":"https://pith.science/api/pith-number/TVQ6NI7HK5VX53IWS7ER7LUEDO/graph.json","fetch_events":"https://pith.science/api/pith-number/TVQ6NI7HK5VX53IWS7ER7LUEDO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TVQ6NI7HK5VX53IWS7ER7LUEDO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TVQ6NI7HK5VX53IWS7ER7LUEDO/action/storage_attestation","attest_author":"https://pith.science/pith/TVQ6NI7HK5VX53IWS7ER7LUEDO/action/author_attestation","sign_citation":"https://pith.science/pith/TVQ6NI7HK5VX53IWS7ER7LUEDO/action/citation_signature","submit_replication":"https://pith.science/pith/TVQ6NI7HK5VX53IWS7ER7LUEDO/action/replication_record"}},"created_at":"2026-07-05T08:05:19.293533+00:00","updated_at":"2026-07-05T08:05:19.293533+00:00"}