{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ZS2WXIWNRD2AJONDZQP6SQX5L2","short_pith_number":"pith:ZS2WXIWN","schema_version":"1.0","canonical_sha256":"ccb56ba2cd88f404b9a3cc1fe942fd5ebbd98e597ae6dd8d4409bf32e8b681fa","source":{"kind":"arxiv","id":"2407.13399","version":3},"attestation_state":"computed","paper":{"title":"Correcting the Mythos of KL-Regularization: Direct Alignment without Overoptimization via Chi-Squared Preference Optimization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.AI","authors_text":"Akshay Krishnamurthy, Audrey Huang, Dylan J. Foster, Jason D. Lee, Tengyang Xie, Wenhao Zhan, Wen Sun","submitted_at":"2024-07-18T11:08:40Z","abstract_excerpt":"Language model alignment methods such as reinforcement learning from human feedback (RLHF) have led to impressive advances in language model capabilities, but are limited by a widely observed phenomenon known as overoptimization, where the quality of the language model degrades over the course of the alignment process. As the model optimizes performance with respect to an offline reward model, it overfits to inaccuracies and drifts away from preferred responses covered by the data. To discourage such distribution shift, KL-regularization is widely employed in existing offline alignment methods"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.13399","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-07-18T11:08:40Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"b68e00937b888249c3e44bc73e138190384bae96fa2fb2416a3b532bbf111ca8","abstract_canon_sha256":"b73f74b81558abf3584ecaa403996a6bd99fb5970dab209d11cfe58b616e1fff"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:16:11.691659Z","signature_b64":"eYffCKLQW0th9fJj+0fwn8whie1gbRJRHn4/F04etXxUPrDe0g58BQkfxXgRWFOve0TmVi9829fzlqZja8NpCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ccb56ba2cd88f404b9a3cc1fe942fd5ebbd98e597ae6dd8d4409bf32e8b681fa","last_reissued_at":"2026-07-05T10:16:11.691134Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:16:11.691134Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Correcting the Mythos of KL-Regularization: Direct Alignment without Overoptimization via Chi-Squared Preference Optimization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.AI","authors_text":"Akshay Krishnamurthy, Audrey Huang, Dylan J. Foster, Jason D. Lee, Tengyang Xie, Wenhao Zhan, Wen Sun","submitted_at":"2024-07-18T11:08:40Z","abstract_excerpt":"Language model alignment methods such as reinforcement learning from human feedback (RLHF) have led to impressive advances in language model capabilities, but are limited by a widely observed phenomenon known as overoptimization, where the quality of the language model degrades over the course of the alignment process. As the model optimizes performance with respect to an offline reward model, it overfits to inaccuracies and drifts away from preferred responses covered by the data. To discourage such distribution shift, KL-regularization is widely employed in existing offline alignment methods"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.13399","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.13399/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.13399","created_at":"2026-07-05T10:16:11.691189+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.13399v3","created_at":"2026-07-05T10:16:11.691189+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.13399","created_at":"2026-07-05T10:16:11.691189+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZS2WXIWNRD2A","created_at":"2026-07-05T10:16:11.691189+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZS2WXIWNRD2AJOND","created_at":"2026-07-05T10:16:11.691189+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZS2WXIWN","created_at":"2026-07-05T10:16:11.691189+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19607","citing_title":"Which Pairs to Compare for LLM Post-Training?","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03065","citing_title":"OGPO: Sample Efficient Full-Finetuning of Generative Control Policies","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2602.06239","citing_title":"Provably avoiding over-optimization in Direct Preference Optimization without knowing the data distribution","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2509.20265","citing_title":"Failure Modes of Maximum Entropy RLHF","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2602.06239","citing_title":"Provably avoiding over-optimization in Direct Preference Optimization without knowing the data distribution","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09214","citing_title":"Fast Rates for Offline Contextual Bandits with Forward-KL Regularization under Single-Policy Concentrability","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06650","citing_title":"Beyond Negative Rollouts: Positive-Only Policy Optimization with Implicit Negative Gradients","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06977","citing_title":"$f$-Divergence Regularized RLHF: Two Tales of Sampling and Unified Analyses","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03065","citing_title":"OGPO: Sample Efficient Full-Finetuning of Generative Control Policies","ref_index":150,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZS2WXIWNRD2AJONDZQP6SQX5L2","json":"https://pith.science/pith/ZS2WXIWNRD2AJONDZQP6SQX5L2.json","graph_json":"https://pith.science/api/pith-number/ZS2WXIWNRD2AJONDZQP6SQX5L2/graph.json","events_json":"https://pith.science/api/pith-number/ZS2WXIWNRD2AJONDZQP6SQX5L2/events.json","paper":"https://pith.science/paper/ZS2WXIWN"},"agent_actions":{"view_html":"https://pith.science/pith/ZS2WXIWNRD2AJONDZQP6SQX5L2","download_json":"https://pith.science/pith/ZS2WXIWNRD2AJONDZQP6SQX5L2.json","view_paper":"https://pith.science/paper/ZS2WXIWN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.13399&json=true","fetch_graph":"https://pith.science/api/pith-number/ZS2WXIWNRD2AJONDZQP6SQX5L2/graph.json","fetch_events":"https://pith.science/api/pith-number/ZS2WXIWNRD2AJONDZQP6SQX5L2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZS2WXIWNRD2AJONDZQP6SQX5L2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZS2WXIWNRD2AJONDZQP6SQX5L2/action/storage_attestation","attest_author":"https://pith.science/pith/ZS2WXIWNRD2AJONDZQP6SQX5L2/action/author_attestation","sign_citation":"https://pith.science/pith/ZS2WXIWNRD2AJONDZQP6SQX5L2/action/citation_signature","submit_replication":"https://pith.science/pith/ZS2WXIWNRD2AJONDZQP6SQX5L2/action/replication_record"}},"created_at":"2026-07-05T10:16:11.691189+00:00","updated_at":"2026-07-05T10:16:11.691189+00:00"}