{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2023:PYCDSKS33CWPGXPQZTU5GRC7S3","short_pith_number":"pith:PYCDSKS3","canonical_record":{"source":{"id":"2312.09244","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-12-14T18:59:04Z","cross_cats_sorted":[],"title_canon_sha256":"ab9920f313ad944c347bf2ed85e2598707334ad891bad6cf207e94c66f0ae623","abstract_canon_sha256":"8fee102c0fabe128d0e775b1e260f9a9adf9b90e6ca601a65d0a19c9860c7d39"},"schema_version":"1.0"},"canonical_sha256":"7e04392a5bd8acf35df0cce9d3445f96eeee1e2aaeeeb630a6bb86c4eb5037cc","source":{"kind":"arxiv","id":"2312.09244","version":3},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2312.09244","created_at":"2026-07-05T08:56:15Z"},{"alias_kind":"arxiv_version","alias_value":"2312.09244v3","created_at":"2026-07-05T08:56:15Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.09244","created_at":"2026-07-05T08:56:15Z"},{"alias_kind":"pith_short_12","alias_value":"PYCDSKS33CWP","created_at":"2026-07-05T08:56:15Z"},{"alias_kind":"pith_short_16","alias_value":"PYCDSKS33CWPGXPQ","created_at":"2026-07-05T08:56:15Z"},{"alias_kind":"pith_short_8","alias_value":"PYCDSKS3","created_at":"2026-07-05T08:56:15Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2023:PYCDSKS33CWPGXPQZTU5GRC7S3","target":"record","payload":{"canonical_record":{"source":{"id":"2312.09244","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-12-14T18:59:04Z","cross_cats_sorted":[],"title_canon_sha256":"ab9920f313ad944c347bf2ed85e2598707334ad891bad6cf207e94c66f0ae623","abstract_canon_sha256":"8fee102c0fabe128d0e775b1e260f9a9adf9b90e6ca601a65d0a19c9860c7d39"},"schema_version":"1.0"},"canonical_sha256":"7e04392a5bd8acf35df0cce9d3445f96eeee1e2aaeeeb630a6bb86c4eb5037cc","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:56:15.816909Z","signature_b64":"6v9c/aROev5bIjyO7UH/YAKZqL6yjHIYZHg9qUWBsvITeHN3R/NBitW1o39j8d3lAHwvo7oCpz3As9jzlNh3CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7e04392a5bd8acf35df0cce9d3445f96eeee1e2aaeeeb630a6bb86c4eb5037cc","last_reissued_at":"2026-07-05T08:56:15.816481Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:56:15.816481Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2312.09244","source_version":3,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:56:15Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"4crcsYoEN955YcQ8Tzzecgm0bKn/pCnR3j1I1xdkDs/BU6zGDGABKj17XK/Bcji5eFqnSz7plizdGx74Fx3yAg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-06T19:31:36.093928Z"},"content_sha256":"b14cc37ee126cb110b91b0a3f8f75081b2362c503c660d749d62e3bbcd4a8d0b","schema_version":"1.0","event_id":"sha256:b14cc37ee126cb110b91b0a3f8f75081b2362c503c660d749d62e3bbcd4a8d0b"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2023:PYCDSKS33CWPGXPQZTU5GRC7S3","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Helping or Herding? Reward Model Ensembles Mitigate but do not Eliminate Reward Hacking","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Adam Fisch, Ahmad Beirami, Alekh Agarwal, Alex D'Amour, Chirag Nagpal, Deepak Ramachandran, Dj Dvijotham, Jacob Eisenstein, Jonathan Berant, Katherine Heller, Peter Shaw, Stephen Pfohl","submitted_at":"2023-12-14T18:59:04Z","abstract_excerpt":"Reward models play a key role in aligning language model applications towards human preferences. However, this setup creates an incentive for the language model to exploit errors in the reward model to achieve high estimated reward, a phenomenon often termed \\emph{reward hacking}. A natural mitigation is to train an ensemble of reward models, aggregating over model outputs to obtain a more robust reward estimate. We explore the application of reward ensembles to alignment at both training time (through reinforcement learning) and inference time (through reranking). First, we show that reward m"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.09244","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.09244/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:56:15Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"nqn7djwBi+CFdilpR2b8M06rL8sPLLG/CxG6t4JGfEbnbJk/A6ZAJiLR71fQQPJ3RgJ0GXJ+pSMS6qxFEn/OCg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-06T19:31:36.096520Z"},"content_sha256":"b988b94c0d6f7d329e49da6ecdf8d58af15696910216613dccf627589a37fdb3","schema_version":"1.0","event_id":"sha256:b988b94c0d6f7d329e49da6ecdf8d58af15696910216613dccf627589a37fdb3"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/PYCDSKS33CWPGXPQZTU5GRC7S3/bundle.json","state_url":"https://pith.science/pith/PYCDSKS33CWPGXPQZTU5GRC7S3/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/PYCDSKS33CWPGXPQZTU5GRC7S3/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-06T19:31:36Z","links":{"resolver":"https://pith.science/pith/PYCDSKS33CWPGXPQZTU5GRC7S3","bundle":"https://pith.science/pith/PYCDSKS33CWPGXPQZTU5GRC7S3/bundle.json","state":"https://pith.science/pith/PYCDSKS33CWPGXPQZTU5GRC7S3/state.json","well_known_bundle":"https://pith.science/.well-known/pith/PYCDSKS33CWPGXPQZTU5GRC7S3/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2023:PYCDSKS33CWPGXPQZTU5GRC7S3","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"8fee102c0fabe128d0e775b1e260f9a9adf9b90e6ca601a65d0a19c9860c7d39","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-12-14T18:59:04Z","title_canon_sha256":"ab9920f313ad944c347bf2ed85e2598707334ad891bad6cf207e94c66f0ae623"},"schema_version":"1.0","source":{"id":"2312.09244","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2312.09244","created_at":"2026-07-05T08:56:15Z"},{"alias_kind":"arxiv_version","alias_value":"2312.09244v3","created_at":"2026-07-05T08:56:15Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.09244","created_at":"2026-07-05T08:56:15Z"},{"alias_kind":"pith_short_12","alias_value":"PYCDSKS33CWP","created_at":"2026-07-05T08:56:15Z"},{"alias_kind":"pith_short_16","alias_value":"PYCDSKS33CWPGXPQ","created_at":"2026-07-05T08:56:15Z"},{"alias_kind":"pith_short_8","alias_value":"PYCDSKS3","created_at":"2026-07-05T08:56:15Z"}],"graph_snapshots":[{"event_id":"sha256:b988b94c0d6f7d329e49da6ecdf8d58af15696910216613dccf627589a37fdb3","target":"graph","created_at":"2026-07-05T08:56:15Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2312.09244/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reward models play a key role in aligning language model applications towards human preferences. However, this setup creates an incentive for the language model to exploit errors in the reward model to achieve high estimated reward, a phenomenon often termed \\emph{reward hacking}. A natural mitigation is to train an ensemble of reward models, aggregating over model outputs to obtain a more robust reward estimate. We explore the application of reward ensembles to alignment at both training time (through reinforcement learning) and inference time (through reranking). First, we show that reward m","authors_text":"Adam Fisch, Ahmad Beirami, Alekh Agarwal, Alex D'Amour, Chirag Nagpal, Deepak Ramachandran, Dj Dvijotham, Jacob Eisenstein, Jonathan Berant, Katherine Heller, Peter Shaw, Stephen Pfohl","cross_cats":[],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-12-14T18:59:04Z","title":"Helping or Herding? Reward Model Ensembles Mitigate but do not Eliminate Reward Hacking"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.09244","kind":"arxiv","version":3},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:b14cc37ee126cb110b91b0a3f8f75081b2362c503c660d749d62e3bbcd4a8d0b","target":"record","created_at":"2026-07-05T08:56:15Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"8fee102c0fabe128d0e775b1e260f9a9adf9b90e6ca601a65d0a19c9860c7d39","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-12-14T18:59:04Z","title_canon_sha256":"ab9920f313ad944c347bf2ed85e2598707334ad891bad6cf207e94c66f0ae623"},"schema_version":"1.0","source":{"id":"2312.09244","kind":"arxiv","version":3}},"canonical_sha256":"7e04392a5bd8acf35df0cce9d3445f96eeee1e2aaeeeb630a6bb86c4eb5037cc","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"7e04392a5bd8acf35df0cce9d3445f96eeee1e2aaeeeb630a6bb86c4eb5037cc","first_computed_at":"2026-07-05T08:56:15.816481Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T08:56:15.816481Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"6v9c/aROev5bIjyO7UH/YAKZqL6yjHIYZHg9qUWBsvITeHN3R/NBitW1o39j8d3lAHwvo7oCpz3As9jzlNh3CA==","signature_status":"signed_v1","signed_at":"2026-07-05T08:56:15.816909Z","signed_message":"canonical_sha256_bytes"},"source_id":"2312.09244","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:b14cc37ee126cb110b91b0a3f8f75081b2362c503c660d749d62e3bbcd4a8d0b","sha256:b988b94c0d6f7d329e49da6ecdf8d58af15696910216613dccf627589a37fdb3"],"state_sha256":"596e8341683fb31fa92ce916599c5bf64b25f45438da6fae4b61caebc424c060"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"UF8HDiWwQ2z+qno7lvL47kmAxRrwv6LoHOi3AHpuKCg5OhjrS1XruOAh0pZ69V/0a201bD05d8uKC9kino8mDA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-06T19:31:36.113728Z","bundle_sha256":"d599d352e2f593be0907a11fc1ff47939d2eabb4d6bd1addb533f1a9da88056c"}}