{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:VPXEM6ESZOGQ25HZXNT2Q2YQZ3","short_pith_number":"pith:VPXEM6ES","schema_version":"1.0","canonical_sha256":"abee467892cb8d0d74f9bb67a86b10ceeffd8e8e430d775e166d757aa19e3c3c","source":{"kind":"arxiv","id":"2403.05006","version":1},"attestation_state":"computed","paper":{"title":"Provable Multi-Party Reinforcement Learning with Diverse Human Feedback","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ME","stat.ML"],"primary_cat":"cs.LG","authors_text":"Huiying Zhong, Linjun Zhang, Weijie J. Su, Zhiwei Steven Wu, Zhun Deng","submitted_at":"2024-03-08T03:05:11Z","abstract_excerpt":"Reinforcement learning with human feedback (RLHF) is an emerging paradigm to align models with human preferences. Typically, RLHF aggregates preferences from multiple individuals who have diverse viewpoints that may conflict with each other. Our work \\textit{initiates} the theoretical study of multi-party RLHF that explicitly models the diverse preferences of multiple individuals. We show how traditional RLHF approaches can fail since learning a single reward function cannot capture and balance the preferences of multiple individuals. To overcome such limitations, we incorporate meta-learning "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.05006","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-03-08T03:05:11Z","cross_cats_sorted":["cs.AI","stat.ME","stat.ML"],"title_canon_sha256":"cda025dc2b254b0d772c92ae3c957b474445365ca095908d0f4e4e58eef68ed4","abstract_canon_sha256":"6ededc216722f56f11acc847d7cf6147e8baf5e965e2eecb840d43d66c5d641c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:53:37.089959Z","signature_b64":"wShuJvk99KTgeT8J1xpGm85hg6JOA8HrPEtjmMh+cB+zTFKQahsMAM1dxU++afVAkCPPF5mH4CpEcJfYXVXSBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"abee467892cb8d0d74f9bb67a86b10ceeffd8e8e430d775e166d757aa19e3c3c","last_reissued_at":"2026-07-05T07:53:37.089401Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:53:37.089401Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Provable Multi-Party Reinforcement Learning with Diverse Human Feedback","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ME","stat.ML"],"primary_cat":"cs.LG","authors_text":"Huiying Zhong, Linjun Zhang, Weijie J. Su, Zhiwei Steven Wu, Zhun Deng","submitted_at":"2024-03-08T03:05:11Z","abstract_excerpt":"Reinforcement learning with human feedback (RLHF) is an emerging paradigm to align models with human preferences. Typically, RLHF aggregates preferences from multiple individuals who have diverse viewpoints that may conflict with each other. Our work \\textit{initiates} the theoretical study of multi-party RLHF that explicitly models the diverse preferences of multiple individuals. We show how traditional RLHF approaches can fail since learning a single reward function cannot capture and balance the preferences of multiple individuals. To overcome such limitations, we incorporate meta-learning "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.05006","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.05006/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.05006","created_at":"2026-07-05T07:53:37.089468+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.05006v1","created_at":"2026-07-05T07:53:37.089468+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.05006","created_at":"2026-07-05T07:53:37.089468+00:00"},{"alias_kind":"pith_short_12","alias_value":"VPXEM6ESZOGQ","created_at":"2026-07-05T07:53:37.089468+00:00"},{"alias_kind":"pith_short_16","alias_value":"VPXEM6ESZOGQ25HZ","created_at":"2026-07-05T07:53:37.089468+00:00"},{"alias_kind":"pith_short_8","alias_value":"VPXEM6ES","created_at":"2026-07-05T07:53:37.089468+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.02507","citing_title":"Reinforcement Learning from Human Feedback: A Statistical Perspective","ref_index":93,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VPXEM6ESZOGQ25HZXNT2Q2YQZ3","json":"https://pith.science/pith/VPXEM6ESZOGQ25HZXNT2Q2YQZ3.json","graph_json":"https://pith.science/api/pith-number/VPXEM6ESZOGQ25HZXNT2Q2YQZ3/graph.json","events_json":"https://pith.science/api/pith-number/VPXEM6ESZOGQ25HZXNT2Q2YQZ3/events.json","paper":"https://pith.science/paper/VPXEM6ES"},"agent_actions":{"view_html":"https://pith.science/pith/VPXEM6ESZOGQ25HZXNT2Q2YQZ3","download_json":"https://pith.science/pith/VPXEM6ESZOGQ25HZXNT2Q2YQZ3.json","view_paper":"https://pith.science/paper/VPXEM6ES","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.05006&json=true","fetch_graph":"https://pith.science/api/pith-number/VPXEM6ESZOGQ25HZXNT2Q2YQZ3/graph.json","fetch_events":"https://pith.science/api/pith-number/VPXEM6ESZOGQ25HZXNT2Q2YQZ3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VPXEM6ESZOGQ25HZXNT2Q2YQZ3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VPXEM6ESZOGQ25HZXNT2Q2YQZ3/action/storage_attestation","attest_author":"https://pith.science/pith/VPXEM6ESZOGQ25HZXNT2Q2YQZ3/action/author_attestation","sign_citation":"https://pith.science/pith/VPXEM6ESZOGQ25HZXNT2Q2YQZ3/action/citation_signature","submit_replication":"https://pith.science/pith/VPXEM6ESZOGQ25HZXNT2Q2YQZ3/action/replication_record"}},"created_at":"2026-07-05T07:53:37.089468+00:00","updated_at":"2026-07-05T07:53:37.089468+00:00"}