{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ZJ2WXQJX44UDFPYCYQJ3UHZF2F","short_pith_number":"pith:ZJ2WXQJX","schema_version":"1.0","canonical_sha256":"ca756bc137e72832bf02c413ba1f25d15826576897240576d27a02338264901c","source":{"kind":"arxiv","id":"2503.11701","version":1},"attestation_state":"computed","paper":{"title":"A Survey of Direct Preference Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Dacheng Tao, Fei Huang, Junjie Zhang, Kongcheng Zhang, Mingli Song, Rongcheng Tu, Shunyu Liu, Ting-En Lin, Wenkai Fang, Yang Zhou, Yongbin Li, Zetian Hu","submitted_at":"2025-03-12T08:45:15Z","abstract_excerpt":"Large Language Models (LLMs) have demonstrated unprecedented generative capabilities, yet their alignment with human values remains critical for ensuring helpful and harmless deployments. While Reinforcement Learning from Human Feedback (RLHF) has emerged as a powerful paradigm for aligning LLMs with human preferences, its reliance on complex reward modeling introduces inherent trade-offs in computational efficiency and training stability. In this context, Direct Preference Optimization (DPO) has recently gained prominence as a streamlined alternative that directly optimizes LLMs using human p"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.11701","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-03-12T08:45:15Z","cross_cats_sorted":[],"title_canon_sha256":"9200ecaf1f762d55301725b9701cf3040f99a22f5b409bf1aff571302d5938ec","abstract_canon_sha256":"c24ba1a6f383579f9db2b5e506906db2c8dba2964f7ab95ab586ffe4e1cab8a5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:31:58.404856Z","signature_b64":"a2mpzx40Pk1dW74hfE4IYlf06l1g+LTX1feQVn5dMM9ZXy/fRTvfmSOc0AwGKqXdGC3ky2OwBbjEAInTHSYXCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ca756bc137e72832bf02c413ba1f25d15826576897240576d27a02338264901c","last_reissued_at":"2026-07-05T10:31:58.404308Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:31:58.404308Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Survey of Direct Preference Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Dacheng Tao, Fei Huang, Junjie Zhang, Kongcheng Zhang, Mingli Song, Rongcheng Tu, Shunyu Liu, Ting-En Lin, Wenkai Fang, Yang Zhou, Yongbin Li, Zetian Hu","submitted_at":"2025-03-12T08:45:15Z","abstract_excerpt":"Large Language Models (LLMs) have demonstrated unprecedented generative capabilities, yet their alignment with human values remains critical for ensuring helpful and harmless deployments. While Reinforcement Learning from Human Feedback (RLHF) has emerged as a powerful paradigm for aligning LLMs with human preferences, its reliance on complex reward modeling introduces inherent trade-offs in computational efficiency and training stability. In this context, Direct Preference Optimization (DPO) has recently gained prominence as a streamlined alternative that directly optimizes LLMs using human p"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.11701","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.11701/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.11701","created_at":"2026-07-05T10:31:58.404386+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.11701v1","created_at":"2026-07-05T10:31:58.404386+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.11701","created_at":"2026-07-05T10:31:58.404386+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZJ2WXQJX44UD","created_at":"2026-07-05T10:31:58.404386+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZJ2WXQJX44UDFPYC","created_at":"2026-07-05T10:31:58.404386+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZJ2WXQJX","created_at":"2026-07-05T10:31:58.404386+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18323","citing_title":"Reliable Neural-Codec Text-to-Speech by ASR Self-Verification and Distillation: Near-Zero Catastrophic Failures Across Models and Codecs","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30930","citing_title":"TUX: Measuring Human--AI Tacit Understanding","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2509.03526","citing_title":"Enhancing Speech Large Language Models through Reinforced Behavior Alignment","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2603.18113","citing_title":"VC-Soup: Value-Consistency Guided Multi-Value Alignment for Large Language Models","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12288","citing_title":"TokenRatio: Principled Token-Level Preference Optimization via Ratio Matching","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12288","citing_title":"TokenRatio: Principled Token-Level Preference Optimization via Ratio Matching","ref_index":177,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11259","citing_title":"Mobile GUI Agent Privacy Personalization with Trajectory Induced Preference Optimization","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07941","citing_title":"Large Language Model Post-Training: A Unified View of Off-Policy and On-Policy Learning","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZJ2WXQJX44UDFPYCYQJ3UHZF2F","json":"https://pith.science/pith/ZJ2WXQJX44UDFPYCYQJ3UHZF2F.json","graph_json":"https://pith.science/api/pith-number/ZJ2WXQJX44UDFPYCYQJ3UHZF2F/graph.json","events_json":"https://pith.science/api/pith-number/ZJ2WXQJX44UDFPYCYQJ3UHZF2F/events.json","paper":"https://pith.science/paper/ZJ2WXQJX"},"agent_actions":{"view_html":"https://pith.science/pith/ZJ2WXQJX44UDFPYCYQJ3UHZF2F","download_json":"https://pith.science/pith/ZJ2WXQJX44UDFPYCYQJ3UHZF2F.json","view_paper":"https://pith.science/paper/ZJ2WXQJX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.11701&json=true","fetch_graph":"https://pith.science/api/pith-number/ZJ2WXQJX44UDFPYCYQJ3UHZF2F/graph.json","fetch_events":"https://pith.science/api/pith-number/ZJ2WXQJX44UDFPYCYQJ3UHZF2F/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZJ2WXQJX44UDFPYCYQJ3UHZF2F/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZJ2WXQJX44UDFPYCYQJ3UHZF2F/action/storage_attestation","attest_author":"https://pith.science/pith/ZJ2WXQJX44UDFPYCYQJ3UHZF2F/action/author_attestation","sign_citation":"https://pith.science/pith/ZJ2WXQJX44UDFPYCYQJ3UHZF2F/action/citation_signature","submit_replication":"https://pith.science/pith/ZJ2WXQJX44UDFPYCYQJ3UHZF2F/action/replication_record"}},"created_at":"2026-07-05T10:31:58.404386+00:00","updated_at":"2026-07-05T10:31:58.404386+00:00"}