{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:BXUE7VOQG4R4L7H775PEWR4CPS","short_pith_number":"pith:BXUE7VOQ","schema_version":"1.0","canonical_sha256":"0de84fd5d03723c5fcffff5e4b47827c9c45489d3266d5a3c91bbda055be55ee","source":{"kind":"arxiv","id":"2405.00254","version":2},"attestation_state":"computed","paper":{"title":"RLHF from Heterogeneous Feedback via Personalization and Preference Aggregation","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Asuman Ozdaglar, Chanwoo Park, Dingwen Kong, Kaiqing Zhang, Mingyang Liu","submitted_at":"2024-04-30T23:57:23Z","abstract_excerpt":"Reinforcement learning from human feedback (RLHF) has been an effective technique for aligning AI systems with human values, with remarkable successes in fine-tuning large-language models recently. Most existing RLHF paradigms make the underlying assumption that human preferences are relatively homogeneous, and can be encoded by a single reward model. In this paper, we focus on addressing the issues due to the inherent heterogeneity in human preferences, as well as their potential strategic behavior in providing feedback. Specifically, we propose two frameworks to address heterogeneous human f"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.00254","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.AI","submitted_at":"2024-04-30T23:57:23Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"ee5d02ec59d823d0d4a4ad7c680212961c7bffeb356874cd840f4c0352b4102c","abstract_canon_sha256":"ebf509ae404865ae07cd71eacfff34203f219771bd57d68fdc75cd9589451d5f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:23:22.077499Z","signature_b64":"TZYP45wJajceksNhzqQYNjFzyKsrkvBEqrV0GllbPex4cSfbtfBy5hFQ8B2TtjAOVlPmx1bg5s4BGtpTqPPqBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0de84fd5d03723c5fcffff5e4b47827c9c45489d3266d5a3c91bbda055be55ee","last_reissued_at":"2026-07-05T08:23:22.077004Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:23:22.077004Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RLHF from Heterogeneous Feedback via Personalization and Preference Aggregation","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Asuman Ozdaglar, Chanwoo Park, Dingwen Kong, Kaiqing Zhang, Mingyang Liu","submitted_at":"2024-04-30T23:57:23Z","abstract_excerpt":"Reinforcement learning from human feedback (RLHF) has been an effective technique for aligning AI systems with human values, with remarkable successes in fine-tuning large-language models recently. Most existing RLHF paradigms make the underlying assumption that human preferences are relatively homogeneous, and can be encoded by a single reward model. In this paper, we focus on addressing the issues due to the inherent heterogeneity in human preferences, as well as their potential strategic behavior in providing feedback. Specifically, we propose two frameworks to address heterogeneous human f"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.00254","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.00254/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.00254","created_at":"2026-07-05T08:23:22.077074+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.00254v2","created_at":"2026-07-05T08:23:22.077074+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.00254","created_at":"2026-07-05T08:23:22.077074+00:00"},{"alias_kind":"pith_short_12","alias_value":"BXUE7VOQG4R4","created_at":"2026-07-05T08:23:22.077074+00:00"},{"alias_kind":"pith_short_16","alias_value":"BXUE7VOQG4R4L7H7","created_at":"2026-07-05T08:23:22.077074+00:00"},{"alias_kind":"pith_short_8","alias_value":"BXUE7VOQ","created_at":"2026-07-05T08:23:22.077074+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01485","citing_title":"CoPersona: Collaborative Persona Graphs for Robust LLM Personalization","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10569","citing_title":"Hidden Consensus:Preference-Validity Compression in Human Feedback","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07988","citing_title":"PAFO: Pareto Fairness Optimization for Personalized Reward Modeling","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07724","citing_title":"Curated Synthetic Data Doesn't Have to Collapse: A Theoretical Study of Generative Retraining with Pluralistic Preferences","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30323","citing_title":"In-Context Reward Adaptation for Robust Preference Modeling","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30903","citing_title":"Inverse Reinforcement Learning without an Optimal Demonstrator: A Feasible Reward Set Approach","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07724","citing_title":"Curated Synthetic Data Doesn't Have to Collapse: A Theoretical Study of Generative Retraining with Pluralistic Preferences","ref_index":117,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BXUE7VOQG4R4L7H775PEWR4CPS","json":"https://pith.science/pith/BXUE7VOQG4R4L7H775PEWR4CPS.json","graph_json":"https://pith.science/api/pith-number/BXUE7VOQG4R4L7H775PEWR4CPS/graph.json","events_json":"https://pith.science/api/pith-number/BXUE7VOQG4R4L7H775PEWR4CPS/events.json","paper":"https://pith.science/paper/BXUE7VOQ"},"agent_actions":{"view_html":"https://pith.science/pith/BXUE7VOQG4R4L7H775PEWR4CPS","download_json":"https://pith.science/pith/BXUE7VOQG4R4L7H775PEWR4CPS.json","view_paper":"https://pith.science/paper/BXUE7VOQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.00254&json=true","fetch_graph":"https://pith.science/api/pith-number/BXUE7VOQG4R4L7H775PEWR4CPS/graph.json","fetch_events":"https://pith.science/api/pith-number/BXUE7VOQG4R4L7H775PEWR4CPS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BXUE7VOQG4R4L7H775PEWR4CPS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BXUE7VOQG4R4L7H775PEWR4CPS/action/storage_attestation","attest_author":"https://pith.science/pith/BXUE7VOQG4R4L7H775PEWR4CPS/action/author_attestation","sign_citation":"https://pith.science/pith/BXUE7VOQG4R4L7H775PEWR4CPS/action/citation_signature","submit_replication":"https://pith.science/pith/BXUE7VOQG4R4L7H775PEWR4CPS/action/replication_record"}},"created_at":"2026-07-05T08:23:22.077074+00:00","updated_at":"2026-07-05T08:23:22.077074+00:00"}