{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:G5I7QT5TYJLTKNJNU5RKPE3LLE","short_pith_number":"pith:G5I7QT5T","schema_version":"1.0","canonical_sha256":"3751f84fb3c25735352da762a7936b593bcf6477daff25017b8a8ac3e372848d","source":{"kind":"arxiv","id":"2402.00856","version":4},"attestation_state":"computed","paper":{"title":"Towards Efficient Exact Optimization of Language Model Alignment","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Cheng Lu, Haozhe Ji, Hongning Wang, Jie Tang, Jun Zhu, Minlie Huang, Pei Ke, Yilin Niu","submitted_at":"2024-02-01T18:51:54Z","abstract_excerpt":"The alignment of language models with human preferences is vital for their application in real-world tasks. The problem is formulated as optimizing the model's policy to maximize the expected reward that reflects human preferences with minimal deviation from the initial policy. While considered as a straightforward solution, reinforcement learning (RL) suffers from high variance in policy updates, which impedes efficient policy improvement. Recently, direct preference optimization (DPO) was proposed to directly optimize the policy from preference data. However, we show that DPO derived based o"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.00856","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-02-01T18:51:54Z","cross_cats_sorted":[],"title_canon_sha256":"239a5e8c03b0fd3faa2a633c06854854ab58450b329d32cb5a756d7a9f428b00","abstract_canon_sha256":"e5518807102ea0cd0ed6e694fc4250e43c0d274f97afbb60ed6709b397374541"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:27:45.724928Z","signature_b64":"KPPSjQtIxfWvEHVzSi4FRhCKdQbUzjD8Li84+aaLn1bRtIpYhF2qNCI1h04cl2/r793HWC18mk8cFZ6D0disAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3751f84fb3c25735352da762a7936b593bcf6477daff25017b8a8ac3e372848d","last_reissued_at":"2026-07-05T08:27:45.724407Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:27:45.724407Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards Efficient Exact Optimization of Language Model Alignment","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Cheng Lu, Haozhe Ji, Hongning Wang, Jie Tang, Jun Zhu, Minlie Huang, Pei Ke, Yilin Niu","submitted_at":"2024-02-01T18:51:54Z","abstract_excerpt":"The alignment of language models with human preferences is vital for their application in real-world tasks. The problem is formulated as optimizing the model's policy to maximize the expected reward that reflects human preferences with minimal deviation from the initial policy. While considered as a straightforward solution, reinforcement learning (RL) suffers from high variance in policy updates, which impedes efficient policy improvement. Recently, direct preference optimization (DPO) was proposed to directly optimize the policy from preference data. However, we show that DPO derived based o"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.00856","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.00856/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.00856","created_at":"2026-07-05T08:27:45.724476+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.00856v4","created_at":"2026-07-05T08:27:45.724476+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.00856","created_at":"2026-07-05T08:27:45.724476+00:00"},{"alias_kind":"pith_short_12","alias_value":"G5I7QT5TYJLT","created_at":"2026-07-05T08:27:45.724476+00:00"},{"alias_kind":"pith_short_16","alias_value":"G5I7QT5TYJLTKNJN","created_at":"2026-07-05T08:27:45.724476+00:00"},{"alias_kind":"pith_short_8","alias_value":"G5I7QT5T","created_at":"2026-07-05T08:27:45.724476+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2509.08827","citing_title":"A Survey of Reinforcement Learning for Large Reasoning Models","ref_index":231,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12112","citing_title":"When Policy Entropy Constraint Fails: Preserving Diversity in Flow-based RLHF via Perceptual Entropy","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11217","citing_title":"Leveraging RAG for Training-Free Alignment of LLMs","ref_index":32,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/G5I7QT5TYJLTKNJNU5RKPE3LLE","json":"https://pith.science/pith/G5I7QT5TYJLTKNJNU5RKPE3LLE.json","graph_json":"https://pith.science/api/pith-number/G5I7QT5TYJLTKNJNU5RKPE3LLE/graph.json","events_json":"https://pith.science/api/pith-number/G5I7QT5TYJLTKNJNU5RKPE3LLE/events.json","paper":"https://pith.science/paper/G5I7QT5T"},"agent_actions":{"view_html":"https://pith.science/pith/G5I7QT5TYJLTKNJNU5RKPE3LLE","download_json":"https://pith.science/pith/G5I7QT5TYJLTKNJNU5RKPE3LLE.json","view_paper":"https://pith.science/paper/G5I7QT5T","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.00856&json=true","fetch_graph":"https://pith.science/api/pith-number/G5I7QT5TYJLTKNJNU5RKPE3LLE/graph.json","fetch_events":"https://pith.science/api/pith-number/G5I7QT5TYJLTKNJNU5RKPE3LLE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/G5I7QT5TYJLTKNJNU5RKPE3LLE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/G5I7QT5TYJLTKNJNU5RKPE3LLE/action/storage_attestation","attest_author":"https://pith.science/pith/G5I7QT5TYJLTKNJNU5RKPE3LLE/action/author_attestation","sign_citation":"https://pith.science/pith/G5I7QT5TYJLTKNJNU5RKPE3LLE/action/citation_signature","submit_replication":"https://pith.science/pith/G5I7QT5TYJLTKNJNU5RKPE3LLE/action/replication_record"}},"created_at":"2026-07-05T08:27:45.724476+00:00","updated_at":"2026-07-05T08:27:45.724476+00:00"}