{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:RHV2BKAKJWN75WQMAE7DUKHNXK","short_pith_number":"pith:RHV2BKAK","schema_version":"1.0","canonical_sha256":"89eba0a80a4d9bfeda0c013e3a28edba80f37ef9a6db430aacaee0059f33d637","source":{"kind":"arxiv","id":"2302.08215","version":2},"attestation_state":"computed","paper":{"title":"Aligning Language Models with Preferences through f-divergence Minimization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG","stat.ML"],"primary_cat":"cs.CL","authors_text":"Dongyoung Go, Germ\\'an Kruszewski, Jos Rozen, Marc Dymetman, Nahyeon Ryu, Tomasz Korbak","submitted_at":"2023-02-16T10:59:39Z","abstract_excerpt":"Aligning language models with preferences can be posed as approximating a target distribution representing some desired behavior. Existing approaches differ both in the functional form of the target distribution and the algorithm used to approximate it. For instance, Reinforcement Learning from Human Feedback (RLHF) corresponds to minimizing a reverse KL from an implicit target distribution arising from a KL penalty in the objective. On the other hand, Generative Distributional Control (GDC) has an explicit target distribution and minimizes a forward KL from it using the Distributional Policy "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2302.08215","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-02-16T10:59:39Z","cross_cats_sorted":["cs.LG","stat.ML"],"title_canon_sha256":"e20b6545b565f5c75cb00bb19562ad5b5c5fcb81582d6eb4f58e62daa9787887","abstract_canon_sha256":"80ea778c6d2d33a444e4e985eea56dc4326ba682cba053baf8d482f4269feb44"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:17:46.424919Z","signature_b64":"varCiR6tTtnzSrebG9JRSacVZcqkfF5hbalZblLSP2TSyoghfCfmcLt+x6OYvwpJJfMPhD628Frhwp0m16UEAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"89eba0a80a4d9bfeda0c013e3a28edba80f37ef9a6db430aacaee0059f33d637","last_reissued_at":"2026-07-05T06:17:46.424473Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:17:46.424473Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Aligning Language Models with Preferences through f-divergence Minimization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG","stat.ML"],"primary_cat":"cs.CL","authors_text":"Dongyoung Go, Germ\\'an Kruszewski, Jos Rozen, Marc Dymetman, Nahyeon Ryu, Tomasz Korbak","submitted_at":"2023-02-16T10:59:39Z","abstract_excerpt":"Aligning language models with preferences can be posed as approximating a target distribution representing some desired behavior. Existing approaches differ both in the functional form of the target distribution and the algorithm used to approximate it. For instance, Reinforcement Learning from Human Feedback (RLHF) corresponds to minimizing a reverse KL from an implicit target distribution arising from a KL penalty in the objective. On the other hand, Generative Distributional Control (GDC) has an explicit target distribution and minimizes a forward KL from it using the Distributional Policy "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2302.08215","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2302.08215/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2302.08215","created_at":"2026-07-05T06:17:46.424531+00:00"},{"alias_kind":"arxiv_version","alias_value":"2302.08215v2","created_at":"2026-07-05T06:17:46.424531+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2302.08215","created_at":"2026-07-05T06:17:46.424531+00:00"},{"alias_kind":"pith_short_12","alias_value":"RHV2BKAKJWN7","created_at":"2026-07-05T06:17:46.424531+00:00"},{"alias_kind":"pith_short_16","alias_value":"RHV2BKAKJWN75WQM","created_at":"2026-07-05T06:17:46.424531+00:00"},{"alias_kind":"pith_short_8","alias_value":"RHV2BKAK","created_at":"2026-07-05T06:17:46.424531+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21498","citing_title":"Balancing Performance and Diversity in GRPO Autoregressive Text-to-Image Post-Training","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12360","citing_title":"Anatomy of Post-Training: Using Interpretability to Characterize Data and Shape the Learning Signal","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15113","citing_title":"Learning from Language Feedback via Variational Policy Distillation","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2310.06987","citing_title":"Catastrophic Jailbreak of Open-source LLMs via Exploiting Generation","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04653","citing_title":"Threshold-Guided Optimization for Visual Generative Models","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06977","citing_title":"$f$-Divergence Regularized RLHF: Two Tales of Sampling and Unified Analyses","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07545","citing_title":"Implicit Preference Alignment for Human Image Animation","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RHV2BKAKJWN75WQMAE7DUKHNXK","json":"https://pith.science/pith/RHV2BKAKJWN75WQMAE7DUKHNXK.json","graph_json":"https://pith.science/api/pith-number/RHV2BKAKJWN75WQMAE7DUKHNXK/graph.json","events_json":"https://pith.science/api/pith-number/RHV2BKAKJWN75WQMAE7DUKHNXK/events.json","paper":"https://pith.science/paper/RHV2BKAK"},"agent_actions":{"view_html":"https://pith.science/pith/RHV2BKAKJWN75WQMAE7DUKHNXK","download_json":"https://pith.science/pith/RHV2BKAKJWN75WQMAE7DUKHNXK.json","view_paper":"https://pith.science/paper/RHV2BKAK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2302.08215&json=true","fetch_graph":"https://pith.science/api/pith-number/RHV2BKAKJWN75WQMAE7DUKHNXK/graph.json","fetch_events":"https://pith.science/api/pith-number/RHV2BKAKJWN75WQMAE7DUKHNXK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RHV2BKAKJWN75WQMAE7DUKHNXK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RHV2BKAKJWN75WQMAE7DUKHNXK/action/storage_attestation","attest_author":"https://pith.science/pith/RHV2BKAKJWN75WQMAE7DUKHNXK/action/author_attestation","sign_citation":"https://pith.science/pith/RHV2BKAKJWN75WQMAE7DUKHNXK/action/citation_signature","submit_replication":"https://pith.science/pith/RHV2BKAKJWN75WQMAE7DUKHNXK/action/replication_record"}},"created_at":"2026-07-05T06:17:46.424531+00:00","updated_at":"2026-07-05T06:17:46.424531+00:00"}