{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ED47JPLPYE7XWMI2AQHRTI2ZQY","short_pith_number":"pith:ED47JPLP","schema_version":"1.0","canonical_sha256":"20f9f4bd6fc13f7b311a040f19a359863f575cde4bcbaf8ce95a7e34585591cc","source":{"kind":"arxiv","id":"2403.08635","version":1},"attestation_state":"computed","paper":{"title":"Human Alignment of Large Language Models through Online Preference Optimisation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Bernardo Avila Pires, Bilal Piot, Charline Le Lan, Daniele Calandriello, Daniel Guo, Mark Rowland, Michal Valko, Pierre Harvey Richemond, Remi Munos, Rishabh Joshi, Tianqi Liu, Yunhao Tang, Zeyu Zheng","submitted_at":"2024-03-13T15:47:26Z","abstract_excerpt":"Ensuring alignment of language models' outputs with human preferences is critical to guarantee a useful, safe, and pleasant user experience. Thus, human alignment has been extensively studied recently and several methods such as Reinforcement Learning from Human Feedback (RLHF), Direct Policy Optimisation (DPO) and Sequence Likelihood Calibration (SLiC) have emerged. In this paper, our contribution is two-fold. First, we show the equivalence between two recent alignment methods, namely Identity Policy Optimisation (IPO) and Nash Mirror Descent (Nash-MD). Second, we introduce a generalisation o"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.08635","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-03-13T15:47:26Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"a53968b4c102eed9f69dbed8ec54c3493188921817cee8b95c7c2b0a8db03c28","abstract_canon_sha256":"78821c512b1de22f458ad2ee179b3ec8b8f6b8e65c3c6a6de5fb12db0fd1a1da"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:55:41.633991Z","signature_b64":"xUjKcdKYwUNsVX01D1d6Lea5vHGf+W5PKlXL1EngWjzjK6LZHhs+VF/0sDDJwaYj0vUDZ673IF4SG2QP3Pd+Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"20f9f4bd6fc13f7b311a040f19a359863f575cde4bcbaf8ce95a7e34585591cc","last_reissued_at":"2026-07-05T07:55:41.633505Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:55:41.633505Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Human Alignment of Large Language Models through Online Preference Optimisation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Bernardo Avila Pires, Bilal Piot, Charline Le Lan, Daniele Calandriello, Daniel Guo, Mark Rowland, Michal Valko, Pierre Harvey Richemond, Remi Munos, Rishabh Joshi, Tianqi Liu, Yunhao Tang, Zeyu Zheng","submitted_at":"2024-03-13T15:47:26Z","abstract_excerpt":"Ensuring alignment of language models' outputs with human preferences is critical to guarantee a useful, safe, and pleasant user experience. Thus, human alignment has been extensively studied recently and several methods such as Reinforcement Learning from Human Feedback (RLHF), Direct Policy Optimisation (DPO) and Sequence Likelihood Calibration (SLiC) have emerged. In this paper, our contribution is two-fold. First, we show the equivalence between two recent alignment methods, namely Identity Policy Optimisation (IPO) and Nash Mirror Descent (Nash-MD). Second, we introduce a generalisation o"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.08635","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.08635/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.08635","created_at":"2026-07-05T07:55:41.633583+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.08635v1","created_at":"2026-07-05T07:55:41.633583+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.08635","created_at":"2026-07-05T07:55:41.633583+00:00"},{"alias_kind":"pith_short_12","alias_value":"ED47JPLPYE7X","created_at":"2026-07-05T07:55:41.633583+00:00"},{"alias_kind":"pith_short_16","alias_value":"ED47JPLPYE7XWMI2","created_at":"2026-07-05T07:55:41.633583+00:00"},{"alias_kind":"pith_short_8","alias_value":"ED47JPLP","created_at":"2026-07-05T07:55:41.633583+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2509.23102","citing_title":"Multiplayer Nash Preference Optimization","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2406.11794","citing_title":"DataComp-LM: In search of the next generation of training sets for language models","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09946","citing_title":"Structure from Strategic Interaction & Uncertainty: Risk Sensitive Games for Robust Preference Learning","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09946","citing_title":"Structure from Strategic Interaction & Uncertainty: Risk Sensitive Games for Robust Preference Learning","ref_index":52,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ED47JPLPYE7XWMI2AQHRTI2ZQY","json":"https://pith.science/pith/ED47JPLPYE7XWMI2AQHRTI2ZQY.json","graph_json":"https://pith.science/api/pith-number/ED47JPLPYE7XWMI2AQHRTI2ZQY/graph.json","events_json":"https://pith.science/api/pith-number/ED47JPLPYE7XWMI2AQHRTI2ZQY/events.json","paper":"https://pith.science/paper/ED47JPLP"},"agent_actions":{"view_html":"https://pith.science/pith/ED47JPLPYE7XWMI2AQHRTI2ZQY","download_json":"https://pith.science/pith/ED47JPLPYE7XWMI2AQHRTI2ZQY.json","view_paper":"https://pith.science/paper/ED47JPLP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.08635&json=true","fetch_graph":"https://pith.science/api/pith-number/ED47JPLPYE7XWMI2AQHRTI2ZQY/graph.json","fetch_events":"https://pith.science/api/pith-number/ED47JPLPYE7XWMI2AQHRTI2ZQY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ED47JPLPYE7XWMI2AQHRTI2ZQY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ED47JPLPYE7XWMI2AQHRTI2ZQY/action/storage_attestation","attest_author":"https://pith.science/pith/ED47JPLPYE7XWMI2AQHRTI2ZQY/action/author_attestation","sign_citation":"https://pith.science/pith/ED47JPLPYE7XWMI2AQHRTI2ZQY/action/citation_signature","submit_replication":"https://pith.science/pith/ED47JPLPYE7XWMI2AQHRTI2ZQY/action/replication_record"}},"created_at":"2026-07-05T07:55:41.633583+00:00","updated_at":"2026-07-05T07:55:41.633583+00:00"}