{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:OHYL4NLGS6Q3HY642D3ZHQI6CA","short_pith_number":"pith:OHYL4NLG","canonical_record":{"source":{"id":"2401.12187","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2024-01-22T18:27:08Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"d0021c503bf6c642c52f047c68f9173c8cbe1bae77444895e494ddd9804d07c0","abstract_canon_sha256":"b867e82676505be3ef92c471c2f0b3299c4cfb8353e1acb692869bc400f10a6b"},"schema_version":"1.0"},"canonical_sha256":"71f0be356697a1b3e3dcd0f793c11e1010abb644aaff43834476de4a513ae7c1","source":{"kind":"arxiv","id":"2401.12187","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2401.12187","created_at":"2026-07-05T07:36:16Z"},{"alias_kind":"arxiv_version","alias_value":"2401.12187v1","created_at":"2026-07-05T07:36:16Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.12187","created_at":"2026-07-05T07:36:16Z"},{"alias_kind":"pith_short_12","alias_value":"OHYL4NLGS6Q3","created_at":"2026-07-05T07:36:16Z"},{"alias_kind":"pith_short_16","alias_value":"OHYL4NLGS6Q3HY64","created_at":"2026-07-05T07:36:16Z"},{"alias_kind":"pith_short_8","alias_value":"OHYL4NLG","created_at":"2026-07-05T07:36:16Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:OHYL4NLGS6Q3HY642D3ZHQI6CA","target":"record","payload":{"canonical_record":{"source":{"id":"2401.12187","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2024-01-22T18:27:08Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"d0021c503bf6c642c52f047c68f9173c8cbe1bae77444895e494ddd9804d07c0","abstract_canon_sha256":"b867e82676505be3ef92c471c2f0b3299c4cfb8353e1acb692869bc400f10a6b"},"schema_version":"1.0"},"canonical_sha256":"71f0be356697a1b3e3dcd0f793c11e1010abb644aaff43834476de4a513ae7c1","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:36:16.646276Z","signature_b64":"fonSFoIUD2Avx6kbPqqJoS5e+nK6zCLjADejx9fDFJNc67PKsM4HiBqsX7ABSWEnvr3hLZFmnQVCDkBUOntRCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"71f0be356697a1b3e3dcd0f793c11e1010abb644aaff43834476de4a513ae7c1","last_reissued_at":"2026-07-05T07:36:16.645788Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:36:16.645788Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2401.12187","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T07:36:16Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"VpO68GCOunaw1/Bi1D/r/PK9QjfZ+fcMOKucKYI5nyaCVC4AbrKvMhBjui7dgTShv6tIxDcZYqebGntqyDchDQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-20T11:17:15.215045Z"},"content_sha256":"6689377a5fd06403ffe7055def7663f6106d4c7ee12429e55d0a1e0e3ea979c7","schema_version":"1.0","event_id":"sha256:6689377a5fd06403ffe7055def7663f6106d4c7ee12429e55d0a1e0e3ea979c7"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:OHYL4NLGS6Q3HY642D3ZHQI6CA","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"WARM: On the Benefits of Weight Averaged Reward Models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Alexandre Ram\\'e, Geoffrey Cideron, Johan Ferret, L\\'eonard Hussenot, Nino Vieillard, Olivier Bachem, Robert Dadashi","submitted_at":"2024-01-22T18:27:08Z","abstract_excerpt":"Aligning large language models (LLMs) with human preferences through reinforcement learning (RLHF) can lead to reward hacking, where LLMs exploit failures in the reward model (RM) to achieve seemingly high rewards without meeting the underlying objectives. We identify two primary challenges when designing RMs to mitigate reward hacking: distribution shifts during the RL process and inconsistencies in human preferences. As a solution, we propose Weight Averaged Reward Models (WARM), first fine-tuning multiple RMs, then averaging them in the weight space. This strategy follows the observation th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.12187","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.12187/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T07:36:16Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"VqvzDIjCS5CZLk+li9iSbnBrkRg+o2M+5818/c1QzJAEPzSSHB2p35d4eXarl66T7wSmTwr0ZY4Vr7Qj/DySBg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-20T11:17:15.215544Z"},"content_sha256":"969966cca847ac090f88cb44a618104860f7402cc28423abb7daf371ac25ac63","schema_version":"1.0","event_id":"sha256:969966cca847ac090f88cb44a618104860f7402cc28423abb7daf371ac25ac63"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/OHYL4NLGS6Q3HY642D3ZHQI6CA/bundle.json","state_url":"https://pith.science/pith/OHYL4NLGS6Q3HY642D3ZHQI6CA/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/OHYL4NLGS6Q3HY642D3ZHQI6CA/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-20T11:17:15Z","links":{"resolver":"https://pith.science/pith/OHYL4NLGS6Q3HY642D3ZHQI6CA","bundle":"https://pith.science/pith/OHYL4NLGS6Q3HY642D3ZHQI6CA/bundle.json","state":"https://pith.science/pith/OHYL4NLGS6Q3HY642D3ZHQI6CA/state.json","well_known_bundle":"https://pith.science/.well-known/pith/OHYL4NLGS6Q3HY642D3ZHQI6CA/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:OHYL4NLGS6Q3HY642D3ZHQI6CA","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"b867e82676505be3ef92c471c2f0b3299c4cfb8353e1acb692869bc400f10a6b","cross_cats_sorted":["cs.AI","cs.CL"],"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2024-01-22T18:27:08Z","title_canon_sha256":"d0021c503bf6c642c52f047c68f9173c8cbe1bae77444895e494ddd9804d07c0"},"schema_version":"1.0","source":{"id":"2401.12187","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2401.12187","created_at":"2026-07-05T07:36:16Z"},{"alias_kind":"arxiv_version","alias_value":"2401.12187v1","created_at":"2026-07-05T07:36:16Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.12187","created_at":"2026-07-05T07:36:16Z"},{"alias_kind":"pith_short_12","alias_value":"OHYL4NLGS6Q3","created_at":"2026-07-05T07:36:16Z"},{"alias_kind":"pith_short_16","alias_value":"OHYL4NLGS6Q3HY64","created_at":"2026-07-05T07:36:16Z"},{"alias_kind":"pith_short_8","alias_value":"OHYL4NLG","created_at":"2026-07-05T07:36:16Z"}],"graph_snapshots":[{"event_id":"sha256:969966cca847ac090f88cb44a618104860f7402cc28423abb7daf371ac25ac63","target":"graph","created_at":"2026-07-05T07:36:16Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2401.12187/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Aligning large language models (LLMs) with human preferences through reinforcement learning (RLHF) can lead to reward hacking, where LLMs exploit failures in the reward model (RM) to achieve seemingly high rewards without meeting the underlying objectives. We identify two primary challenges when designing RMs to mitigate reward hacking: distribution shifts during the RL process and inconsistencies in human preferences. As a solution, we propose Weight Averaged Reward Models (WARM), first fine-tuning multiple RMs, then averaging them in the weight space. This strategy follows the observation th","authors_text":"Alexandre Ram\\'e, Geoffrey Cideron, Johan Ferret, L\\'eonard Hussenot, Nino Vieillard, Olivier Bachem, Robert Dadashi","cross_cats":["cs.AI","cs.CL"],"headline":"","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2024-01-22T18:27:08Z","title":"WARM: On the Benefits of Weight Averaged Reward Models"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.12187","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:6689377a5fd06403ffe7055def7663f6106d4c7ee12429e55d0a1e0e3ea979c7","target":"record","created_at":"2026-07-05T07:36:16Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"b867e82676505be3ef92c471c2f0b3299c4cfb8353e1acb692869bc400f10a6b","cross_cats_sorted":["cs.AI","cs.CL"],"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2024-01-22T18:27:08Z","title_canon_sha256":"d0021c503bf6c642c52f047c68f9173c8cbe1bae77444895e494ddd9804d07c0"},"schema_version":"1.0","source":{"id":"2401.12187","kind":"arxiv","version":1}},"canonical_sha256":"71f0be356697a1b3e3dcd0f793c11e1010abb644aaff43834476de4a513ae7c1","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"71f0be356697a1b3e3dcd0f793c11e1010abb644aaff43834476de4a513ae7c1","first_computed_at":"2026-07-05T07:36:16.645788Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T07:36:16.645788Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"fonSFoIUD2Avx6kbPqqJoS5e+nK6zCLjADejx9fDFJNc67PKsM4HiBqsX7ABSWEnvr3hLZFmnQVCDkBUOntRCA==","signature_status":"signed_v1","signed_at":"2026-07-05T07:36:16.646276Z","signed_message":"canonical_sha256_bytes"},"source_id":"2401.12187","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:6689377a5fd06403ffe7055def7663f6106d4c7ee12429e55d0a1e0e3ea979c7","sha256:969966cca847ac090f88cb44a618104860f7402cc28423abb7daf371ac25ac63"],"state_sha256":"3d5a7125d05f992f7cd1f526b07b147175178c6533377ceb4d94d7d40b19dda9"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"zXufgbjjimsJT+CeZxCORuBUX396oE1ZuBjc38BnrPMNqkalkc5M7cihICGVX2xW82kUD3MLrwJBFYcVnCIRCA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-20T11:17:15.219363Z","bundle_sha256":"b717a2b03a82d9b0017f04e97735e99c388440e8e579499201c8e7a3ffe4d77e"}}