{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:P53UNYGIV2O4ZVHYFY66WVDXBZ","short_pith_number":"pith:P53UNYGI","schema_version":"1.0","canonical_sha256":"7f7746e0c8ae9dccd4f82e3deb54770e767a3892789286a7ba86b30412ced4de","source":{"kind":"arxiv","id":"2408.10075","version":1},"attestation_state":"computed","paper":{"title":"Personalizing Reinforcement Learning from Human Feedback with Variational Preference Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.RO"],"primary_cat":"cs.LG","authors_text":"Abhishek Gupta, Hamish Ivison, Natasha Jaques, Sriyash Poddar, Yanming Wan","submitted_at":"2024-08-19T15:18:30Z","abstract_excerpt":"Reinforcement Learning from Human Feedback (RLHF) is a powerful paradigm for aligning foundation models to human values and preferences. However, current RLHF techniques cannot account for the naturally occurring differences in individual human preferences across a diverse population. When these differences arise, traditional RLHF frameworks simply average over them, leading to inaccurate rewards and poor performance for individual subgroups. To address the need for pluralistic alignment, we develop a class of multimodal RLHF methods. Our proposed techniques are based on a latent variable form"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.10075","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-08-19T15:18:30Z","cross_cats_sorted":["cs.AI","cs.CL","cs.RO"],"title_canon_sha256":"62b36c3662a8989bae5278dcb1defce990f44172d2534b0fec24a9f34a7cfc40","abstract_canon_sha256":"3c838fcaf55d46072f05c60175663dbb057b252d98f82d06dec8bfdd1a891045"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:56:50.997814Z","signature_b64":"MjOoU00W34S9hYDKFYeaONLMMxVLcjKBegKQMUN5jx7YxQpwvvIbw2a1JG2M2wsomOePswOZUTIXk6zAIAs9Cw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7f7746e0c8ae9dccd4f82e3deb54770e767a3892789286a7ba86b30412ced4de","last_reissued_at":"2026-07-05T08:56:50.997437Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:56:50.997437Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Personalizing Reinforcement Learning from Human Feedback with Variational Preference Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.RO"],"primary_cat":"cs.LG","authors_text":"Abhishek Gupta, Hamish Ivison, Natasha Jaques, Sriyash Poddar, Yanming Wan","submitted_at":"2024-08-19T15:18:30Z","abstract_excerpt":"Reinforcement Learning from Human Feedback (RLHF) is a powerful paradigm for aligning foundation models to human values and preferences. However, current RLHF techniques cannot account for the naturally occurring differences in individual human preferences across a diverse population. When these differences arise, traditional RLHF frameworks simply average over them, leading to inaccurate rewards and poor performance for individual subgroups. To address the need for pluralistic alignment, we develop a class of multimodal RLHF methods. Our proposed techniques are based on a latent variable form"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.10075","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.10075/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.10075","created_at":"2026-07-05T08:56:50.997492+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.10075v1","created_at":"2026-07-05T08:56:50.997492+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.10075","created_at":"2026-07-05T08:56:50.997492+00:00"},{"alias_kind":"pith_short_12","alias_value":"P53UNYGIV2O4","created_at":"2026-07-05T08:56:50.997492+00:00"},{"alias_kind":"pith_short_16","alias_value":"P53UNYGIV2O4ZVHY","created_at":"2026-07-05T08:56:50.997492+00:00"},{"alias_kind":"pith_short_8","alias_value":"P53UNYGI","created_at":"2026-07-05T08:56:50.997492+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.17657","citing_title":"Using Cognitive Models to Improve Language Model Simulation of Human Persuasion Games","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02300","citing_title":"Beyond Isolated Behaviors: Hierarchical User Modeling for LLM Personalization","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30873","citing_title":"Federated Variational Preference Alignment with Gumbel-Softmax Prior for Personalized User Preferences","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2412.08812","citing_title":"Test-Time Alignment via Hypothesis Reweighting","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09876","citing_title":"Efficient Personalization of Generative User Interfaces","ref_index":80,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/P53UNYGIV2O4ZVHYFY66WVDXBZ","json":"https://pith.science/pith/P53UNYGIV2O4ZVHYFY66WVDXBZ.json","graph_json":"https://pith.science/api/pith-number/P53UNYGIV2O4ZVHYFY66WVDXBZ/graph.json","events_json":"https://pith.science/api/pith-number/P53UNYGIV2O4ZVHYFY66WVDXBZ/events.json","paper":"https://pith.science/paper/P53UNYGI"},"agent_actions":{"view_html":"https://pith.science/pith/P53UNYGIV2O4ZVHYFY66WVDXBZ","download_json":"https://pith.science/pith/P53UNYGIV2O4ZVHYFY66WVDXBZ.json","view_paper":"https://pith.science/paper/P53UNYGI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.10075&json=true","fetch_graph":"https://pith.science/api/pith-number/P53UNYGIV2O4ZVHYFY66WVDXBZ/graph.json","fetch_events":"https://pith.science/api/pith-number/P53UNYGIV2O4ZVHYFY66WVDXBZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/P53UNYGIV2O4ZVHYFY66WVDXBZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/P53UNYGIV2O4ZVHYFY66WVDXBZ/action/storage_attestation","attest_author":"https://pith.science/pith/P53UNYGIV2O4ZVHYFY66WVDXBZ/action/author_attestation","sign_citation":"https://pith.science/pith/P53UNYGIV2O4ZVHYFY66WVDXBZ/action/citation_signature","submit_replication":"https://pith.science/pith/P53UNYGIV2O4ZVHYFY66WVDXBZ/action/replication_record"}},"created_at":"2026-07-05T08:56:50.997492+00:00","updated_at":"2026-07-05T08:56:50.997492+00:00"}