{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:WCYOYPRQEDEZ5WZDJ3VDKY4R7N","short_pith_number":"pith:WCYOYPRQ","schema_version":"1.0","canonical_sha256":"b0b0ec3e3020c99edb234eea356391fb52b2a49f6f72bd41edc78c2cc483f8ff","source":{"kind":"arxiv","id":"2310.02743","version":2},"attestation_state":"computed","paper":{"title":"Reward Model Ensembles Help Mitigate Overoptimization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"David Krueger, Robert Kirk, Thomas Coste, Usman Anwar","submitted_at":"2023-10-04T11:34:22Z","abstract_excerpt":"Reinforcement learning from human feedback (RLHF) is a standard approach for fine-tuning large language models to follow instructions. As part of this process, learned reward models are used to approximately model human preferences. However, as imperfect representations of the \"true\" reward, these learned reward models are susceptible to overoptimization. Gao et al. (2023) studied this phenomenon in a synthetic human feedback setup with a significantly larger \"gold\" reward model acting as the true reward (instead of humans) and showed that overoptimization remains a persistent problem regardle"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.02743","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-10-04T11:34:22Z","cross_cats_sorted":[],"title_canon_sha256":"22814d6a9c05683e758f9a248dcf493e5df4e87d1ab1b3b5c992955c7823bc66","abstract_canon_sha256":"2f87915085338a6513ba5ca056b89c1f8c5a1a5e5cca6b29ceb7232990fab219"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:54:04.454541Z","signature_b64":"UP4V8LNr8RSgLodt9y7Z66byt+OtEdiEj4fkERss3add7qPoX4H4JZrqbqwOpZP+U08DU1ULER3uM/yyj1IDBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b0b0ec3e3020c99edb234eea356391fb52b2a49f6f72bd41edc78c2cc483f8ff","last_reissued_at":"2026-07-05T07:54:04.454145Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:54:04.454145Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Reward Model Ensembles Help Mitigate Overoptimization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"David Krueger, Robert Kirk, Thomas Coste, Usman Anwar","submitted_at":"2023-10-04T11:34:22Z","abstract_excerpt":"Reinforcement learning from human feedback (RLHF) is a standard approach for fine-tuning large language models to follow instructions. As part of this process, learned reward models are used to approximately model human preferences. However, as imperfect representations of the \"true\" reward, these learned reward models are susceptible to overoptimization. Gao et al. (2023) studied this phenomenon in a synthetic human feedback setup with a significantly larger \"gold\" reward model acting as the true reward (instead of humans) and showed that overoptimization remains a persistent problem regardle"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.02743","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.02743/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.02743","created_at":"2026-07-05T07:54:04.454200+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.02743v2","created_at":"2026-07-05T07:54:04.454200+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.02743","created_at":"2026-07-05T07:54:04.454200+00:00"},{"alias_kind":"pith_short_12","alias_value":"WCYOYPRQEDEZ","created_at":"2026-07-05T07:54:04.454200+00:00"},{"alias_kind":"pith_short_16","alias_value":"WCYOYPRQEDEZ5WZD","created_at":"2026-07-05T07:54:04.454200+00:00"},{"alias_kind":"pith_short_8","alias_value":"WCYOYPRQ","created_at":"2026-07-05T07:54:04.454200+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":19,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05904","citing_title":"More Convincing, Not More Correct: Self-Play Reward Hacking of Reference-Free LLM Judges","ref_index":3,"is_internal_anchor":true},{"citing_arxiv_id":"2606.27291","citing_title":"Designing Reward Signals for Portable Query Generation: A Case Study in Industrial Semantic Job Search","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23597","citing_title":"Against Proxy Optimization","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19818","citing_title":"Uncertainty-Aware Reward Modeling for Stable RLHF","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09711","citing_title":"Proxy Reward Internalization and Mechanistic Exploitation: A Learned Precursor to Reward Hacking and Its Generalization","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21602","citing_title":"Benchmarking and Improving Monitors for Out-Of-Distribution Alignment Failure in LLMs","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29834","citing_title":"STEAM: Self-Supervised Temporal Ensemble Advantage Modeling for Real-World Robot Learning","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21602","citing_title":"Benchmarking and Improving Monitors for Out-Of-Distribution Alignment Failure in LLMs","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21602","citing_title":"Benchmarking and Improving Monitors for Out-Of-Distribution Alignment Failure in LLMs","ref_index":72,"is_internal_anchor":false},{"citing_arxiv_id":"2601.21350","citing_title":"Factored Causal Representation Learning for Robust Reward Modeling in RLHF","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2602.06239","citing_title":"Provably avoiding over-optimization in Direct Preference Optimization without knowing the data distribution","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2602.06239","citing_title":"Provably avoiding over-optimization in Direct Preference Optimization without knowing the data distribution","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10983","citing_title":"TMPO: Trajectory Matching Policy Optimization for Diverse and Efficient Diffusion Alignment","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10983","citing_title":"TMPO: Trajectory Matching Policy Optimization for Diverse and Efficient Diffusion Alignment","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11865","citing_title":"Variance-aware Reward Modeling with Anchor Guidance","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09808","citing_title":"Quantifying the Utility of User Simulators for Building Collaborative LLM Assistants","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18547","citing_title":"FUSE: Ensembling Verifiers with Zero Labeled Data","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07105","citing_title":"Theoretical Limits of Language Model Alignment","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14265","citing_title":"Reinforcement Learning via Value Gradient Flow","ref_index":11,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WCYOYPRQEDEZ5WZDJ3VDKY4R7N","json":"https://pith.science/pith/WCYOYPRQEDEZ5WZDJ3VDKY4R7N.json","graph_json":"https://pith.science/api/pith-number/WCYOYPRQEDEZ5WZDJ3VDKY4R7N/graph.json","events_json":"https://pith.science/api/pith-number/WCYOYPRQEDEZ5WZDJ3VDKY4R7N/events.json","paper":"https://pith.science/paper/WCYOYPRQ"},"agent_actions":{"view_html":"https://pith.science/pith/WCYOYPRQEDEZ5WZDJ3VDKY4R7N","download_json":"https://pith.science/pith/WCYOYPRQEDEZ5WZDJ3VDKY4R7N.json","view_paper":"https://pith.science/paper/WCYOYPRQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.02743&json=true","fetch_graph":"https://pith.science/api/pith-number/WCYOYPRQEDEZ5WZDJ3VDKY4R7N/graph.json","fetch_events":"https://pith.science/api/pith-number/WCYOYPRQEDEZ5WZDJ3VDKY4R7N/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WCYOYPRQEDEZ5WZDJ3VDKY4R7N/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WCYOYPRQEDEZ5WZDJ3VDKY4R7N/action/storage_attestation","attest_author":"https://pith.science/pith/WCYOYPRQEDEZ5WZDJ3VDKY4R7N/action/author_attestation","sign_citation":"https://pith.science/pith/WCYOYPRQEDEZ5WZDJ3VDKY4R7N/action/citation_signature","submit_replication":"https://pith.science/pith/WCYOYPRQEDEZ5WZDJ3VDKY4R7N/action/replication_record"}},"created_at":"2026-07-05T07:54:04.454200+00:00","updated_at":"2026-07-05T07:54:04.454200+00:00"}