{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2023:WCYOYPRQEDEZ5WZDJ3VDKY4R7N","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"2f87915085338a6513ba5ca056b89c1f8c5a1a5e5cca6b29ceb7232990fab219","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-10-04T11:34:22Z","title_canon_sha256":"22814d6a9c05683e758f9a248dcf493e5df4e87d1ab1b3b5c992955c7823bc66"},"schema_version":"1.0","source":{"id":"2310.02743","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2310.02743","created_at":"2026-07-05T07:54:04Z"},{"alias_kind":"arxiv_version","alias_value":"2310.02743v2","created_at":"2026-07-05T07:54:04Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.02743","created_at":"2026-07-05T07:54:04Z"},{"alias_kind":"pith_short_12","alias_value":"WCYOYPRQEDEZ","created_at":"2026-07-05T07:54:04Z"},{"alias_kind":"pith_short_16","alias_value":"WCYOYPRQEDEZ5WZD","created_at":"2026-07-05T07:54:04Z"},{"alias_kind":"pith_short_8","alias_value":"WCYOYPRQ","created_at":"2026-07-05T07:54:04Z"}],"graph_snapshots":[{"event_id":"sha256:5ba6dab8ff9558ebd53dc11ff0ea5a6816d5b4a77bff31f4764f08c767c9475c","target":"graph","created_at":"2026-07-05T07:54:04Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2310.02743/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement learning from human feedback (RLHF) is a standard approach for fine-tuning large language models to follow instructions. As part of this process, learned reward models are used to approximately model human preferences. However, as imperfect representations of the \"true\" reward, these learned reward models are susceptible to overoptimization. Gao et al. (2023) studied this phenomenon in a synthetic human feedback setup with a significantly larger \"gold\" reward model acting as the true reward (instead of humans) and showed that overoptimization remains a persistent problem regardle","authors_text":"David Krueger, Robert Kirk, Thomas Coste, Usman Anwar","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-10-04T11:34:22Z","title":"Reward Model Ensembles Help Mitigate Overoptimization"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.02743","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:d0f5d2a7bca43baec38a68c3e1b3ae44b406e199824ee433c9b7d8d9b9dfe769","target":"record","created_at":"2026-07-05T07:54:04Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"2f87915085338a6513ba5ca056b89c1f8c5a1a5e5cca6b29ceb7232990fab219","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-10-04T11:34:22Z","title_canon_sha256":"22814d6a9c05683e758f9a248dcf493e5df4e87d1ab1b3b5c992955c7823bc66"},"schema_version":"1.0","source":{"id":"2310.02743","kind":"arxiv","version":2}},"canonical_sha256":"b0b0ec3e3020c99edb234eea356391fb52b2a49f6f72bd41edc78c2cc483f8ff","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"b0b0ec3e3020c99edb234eea356391fb52b2a49f6f72bd41edc78c2cc483f8ff","first_computed_at":"2026-07-05T07:54:04.454145Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T07:54:04.454145Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"UP4V8LNr8RSgLodt9y7Z66byt+OtEdiEj4fkERss3add7qPoX4H4JZrqbqwOpZP+U08DU1ULER3uM/yyj1IDBA==","signature_status":"signed_v1","signed_at":"2026-07-05T07:54:04.454541Z","signed_message":"canonical_sha256_bytes"},"source_id":"2310.02743","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:d0f5d2a7bca43baec38a68c3e1b3ae44b406e199824ee433c9b7d8d9b9dfe769","sha256:5ba6dab8ff9558ebd53dc11ff0ea5a6816d5b4a77bff31f4764f08c767c9475c"],"state_sha256":"1a8649fb4ed94a76ea2f9c7eca70a6d627f6b4992529348c281cd18423791322"}