{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:ZAFFDLPL77N6PTXWHCWF2AUGJT","short_pith_number":"pith:ZAFFDLPL","schema_version":"1.0","canonical_sha256":"c80a51adebffdbe7cef638ac5d02864ce33e0515f5499dfba33b6cbe16f09e88","source":{"kind":"arxiv","id":"2309.16240","version":1},"attestation_state":"computed","paper":{"title":"Beyond Reverse KL: Generalizing Direct Preference Optimization with Diverse Divergence Constraints","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Chaoqi Wang, Chenghao Yang, Han Liu, Yibo Jiang, Yuxin Chen","submitted_at":"2023-09-28T08:29:44Z","abstract_excerpt":"The increasing capabilities of large language models (LLMs) raise opportunities for artificial general intelligence but concurrently amplify safety concerns, such as potential misuse of AI systems, necessitating effective AI alignment. Reinforcement Learning from Human Feedback (RLHF) has emerged as a promising pathway towards AI alignment but brings forth challenges due to its complexity and dependence on a separate reward model. Direct Preference Optimization (DPO) has been proposed as an alternative, and it remains equivalent to RLHF under the reverse KL regularization constraint. This pape"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.16240","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-09-28T08:29:44Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"33c0fd9f5ef49d921ed1d0a040eaa17fa2c4c0ab4445cc030944175c4bd06be1","abstract_canon_sha256":"49891ab733a20f0552b81d96d08f006b7d04531dded59ecba8145dcbec8c9fed"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:55:13.185820Z","signature_b64":"8OY06xGJkjmihJnNbgefSnmdQvEtOEC2fyd49fvPJWiuxXi19ZLjTlz8mwJ1otl5ktksfNfFMfdGPEtZ1znDAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c80a51adebffdbe7cef638ac5d02864ce33e0515f5499dfba33b6cbe16f09e88","last_reissued_at":"2026-07-05T06:55:13.185420Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:55:13.185420Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Beyond Reverse KL: Generalizing Direct Preference Optimization with Diverse Divergence Constraints","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Chaoqi Wang, Chenghao Yang, Han Liu, Yibo Jiang, Yuxin Chen","submitted_at":"2023-09-28T08:29:44Z","abstract_excerpt":"The increasing capabilities of large language models (LLMs) raise opportunities for artificial general intelligence but concurrently amplify safety concerns, such as potential misuse of AI systems, necessitating effective AI alignment. Reinforcement Learning from Human Feedback (RLHF) has emerged as a promising pathway towards AI alignment but brings forth challenges due to its complexity and dependence on a separate reward model. Direct Preference Optimization (DPO) has been proposed as an alternative, and it remains equivalent to RLHF under the reverse KL regularization constraint. This pape"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.16240","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.16240/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.16240","created_at":"2026-07-05T06:55:13.185481+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.16240v1","created_at":"2026-07-05T06:55:13.185481+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.16240","created_at":"2026-07-05T06:55:13.185481+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZAFFDLPL77N6","created_at":"2026-07-05T06:55:13.185481+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZAFFDLPL77N6PTXW","created_at":"2026-07-05T06:55:13.185481+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZAFFDLPL","created_at":"2026-07-05T06:55:13.185481+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":15,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24937","citing_title":"The Hitchhiker's Guide to Agentic AI: From Foundations to Systems","ref_index":189,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05800","citing_title":"SALT: When More Rollouts Don't Help in Group-Based Policy Optimization and How to Make Them Matter","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00306","citing_title":"Rethinking the Role of Temperature in Large Language Model Distillation","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2410.19471","citing_title":"Improving Inverse Folding for Peptide Design with Diversity-regularized Direct Preference Optimization","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06139","citing_title":"Listwise Policy Optimization: Group-based RLVR as Target-Projection on the LLM Response Simplex","ref_index":73,"is_internal_anchor":false},{"citing_arxiv_id":"2509.23102","citing_title":"Multiplayer Nash Preference Optimization","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12288","citing_title":"TokenRatio: Principled Token-Level Preference Optimization via Ratio Matching","ref_index":117,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11726","citing_title":"Block-R1: Rethinking the Role of Block Size in Multi-domain Reinforcement Learning for Diffusion Large Language Models","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11726","citing_title":"Block-R1: Rethinking the Role of Block Size in Multi-domain Reinforcement Learning for Diffusion Large Language Models","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12288","citing_title":"TokenRatio: Principled Token-Level Preference Optimization via Ratio Matching","ref_index":117,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09214","citing_title":"Fast Rates for Offline Contextual Bandits with Forward-KL Regularization under Single-Policy Concentrability","ref_index":109,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00195","citing_title":"Diversity in Large Language Models under Supervised Fine-Tuning","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06139","citing_title":"Listwise Policy Optimization: Group-based RLVR as Target-Projection on the LLM Response Simplex","ref_index":72,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06977","citing_title":"$f$-Divergence Regularized RLHF: Two Tales of Sampling and Unified Analyses","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00195","citing_title":"Diversity in Large Language Models under Supervised Fine-Tuning","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZAFFDLPL77N6PTXWHCWF2AUGJT","json":"https://pith.science/pith/ZAFFDLPL77N6PTXWHCWF2AUGJT.json","graph_json":"https://pith.science/api/pith-number/ZAFFDLPL77N6PTXWHCWF2AUGJT/graph.json","events_json":"https://pith.science/api/pith-number/ZAFFDLPL77N6PTXWHCWF2AUGJT/events.json","paper":"https://pith.science/paper/ZAFFDLPL"},"agent_actions":{"view_html":"https://pith.science/pith/ZAFFDLPL77N6PTXWHCWF2AUGJT","download_json":"https://pith.science/pith/ZAFFDLPL77N6PTXWHCWF2AUGJT.json","view_paper":"https://pith.science/paper/ZAFFDLPL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.16240&json=true","fetch_graph":"https://pith.science/api/pith-number/ZAFFDLPL77N6PTXWHCWF2AUGJT/graph.json","fetch_events":"https://pith.science/api/pith-number/ZAFFDLPL77N6PTXWHCWF2AUGJT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZAFFDLPL77N6PTXWHCWF2AUGJT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZAFFDLPL77N6PTXWHCWF2AUGJT/action/storage_attestation","attest_author":"https://pith.science/pith/ZAFFDLPL77N6PTXWHCWF2AUGJT/action/author_attestation","sign_citation":"https://pith.science/pith/ZAFFDLPL77N6PTXWHCWF2AUGJT/action/citation_signature","submit_replication":"https://pith.science/pith/ZAFFDLPL77N6PTXWHCWF2AUGJT/action/replication_record"}},"created_at":"2026-07-05T06:55:13.185481+00:00","updated_at":"2026-07-05T06:55:13.185481+00:00"}