{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2023:RIUB5RACBJQL7EGEVUVT5YPBII","short_pith_number":"pith:RIUB5RAC","canonical_record":{"source":{"id":"2401.00243","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-12-30T14:14:14Z","cross_cats_sorted":[],"title_canon_sha256":"44baa034d026023c1e16ebb715568d6d2f8f0eda6145e1b5bb757b7fbc9664ba","abstract_canon_sha256":"5119e51677a320cff9580c51b4f4191f723589b0d67b6f4286902afb3ae964e9"},"schema_version":"1.0"},"canonical_sha256":"8a281ec4020a60bf90c4ad2b3ee1e14237a643a9b70bcdc0e52a33af97e99329","source":{"kind":"arxiv","id":"2401.00243","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2401.00243","created_at":"2026-07-05T07:29:02Z"},{"alias_kind":"arxiv_version","alias_value":"2401.00243v1","created_at":"2026-07-05T07:29:02Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.00243","created_at":"2026-07-05T07:29:02Z"},{"alias_kind":"pith_short_12","alias_value":"RIUB5RACBJQL","created_at":"2026-07-05T07:29:02Z"},{"alias_kind":"pith_short_16","alias_value":"RIUB5RACBJQL7EGE","created_at":"2026-07-05T07:29:02Z"},{"alias_kind":"pith_short_8","alias_value":"RIUB5RAC","created_at":"2026-07-05T07:29:02Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2023:RIUB5RACBJQL7EGEVUVT5YPBII","target":"record","payload":{"canonical_record":{"source":{"id":"2401.00243","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-12-30T14:14:14Z","cross_cats_sorted":[],"title_canon_sha256":"44baa034d026023c1e16ebb715568d6d2f8f0eda6145e1b5bb757b7fbc9664ba","abstract_canon_sha256":"5119e51677a320cff9580c51b4f4191f723589b0d67b6f4286902afb3ae964e9"},"schema_version":"1.0"},"canonical_sha256":"8a281ec4020a60bf90c4ad2b3ee1e14237a643a9b70bcdc0e52a33af97e99329","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:29:02.710299Z","signature_b64":"3S7zgy+zT/urUdEvh0y6Ems2EoEV600xtqTvbroh1O6FDkyIWgPR4uWBVr1EfKqsJbigHMPU79j1MVC1uB7+CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8a281ec4020a60bf90c4ad2b3ee1e14237a643a9b70bcdc0e52a33af97e99329","last_reissued_at":"2026-07-05T07:29:02.709690Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:29:02.709690Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2401.00243","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T07:29:02Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"KrIKgP/sgq8g5ltT2msTR8L0E2s9ZVe1cb2PnksHiTi5RRHflSiGMogZZfFM3DsMfMU+P6TMOyewbKR7EH9rAQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-07-29T09:29:20.310933Z"},"content_sha256":"1a2dec4b4e8409c13707b2f8476d267cafe8125a56a70e8ef6891be5cb689aac","schema_version":"1.0","event_id":"sha256:1a2dec4b4e8409c13707b2f8476d267cafe8125a56a70e8ef6891be5cb689aac"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2023:RIUB5RACBJQL7EGEVUVT5YPBII","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Uncertainty-Penalized Reinforcement Learning from Human Feedback with Diverse Reward LoRA Ensembles","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Bo Ding, Dawei Feng, Han Zhang, Huaimin Wang, Kele Xu, Yuanzhao Zhai, Yue Yu, Yu Lei","submitted_at":"2023-12-30T14:14:14Z","abstract_excerpt":"Reinforcement learning from human feedback (RLHF) emerges as a promising paradigm for aligning large language models (LLMs). However, a notable challenge in RLHF is overoptimization, where beyond a certain threshold, the pursuit of higher rewards leads to a decline in human preferences. In this paper, we observe the weakness of KL regularization which is commonly employed in existing RLHF methods to address overoptimization. To mitigate this limitation, we scrutinize the RLHF objective in the offline dataset and propose uncertainty-penalized RLHF (UP-RLHF), which incorporates uncertainty regul"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.00243","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.00243/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T07:29:02Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"xPGAbDR7K7x+qs16T+bmoTDMqbz8xmLLhjxVave9iwnGHZsggZ4ppjIMXS9q2SVklnzKSQLgMe0h3qaUNFGRCw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-07-29T09:29:20.311332Z"},"content_sha256":"ef50553b775c541b34ed693deaeaf25dcc774c115e2246f651d9d9889f77fe39","schema_version":"1.0","event_id":"sha256:ef50553b775c541b34ed693deaeaf25dcc774c115e2246f651d9d9889f77fe39"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/RIUB5RACBJQL7EGEVUVT5YPBII/bundle.json","state_url":"https://pith.science/pith/RIUB5RACBJQL7EGEVUVT5YPBII/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/RIUB5RACBJQL7EGEVUVT5YPBII/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-07-29T09:29:20Z","links":{"resolver":"https://pith.science/pith/RIUB5RACBJQL7EGEVUVT5YPBII","bundle":"https://pith.science/pith/RIUB5RACBJQL7EGEVUVT5YPBII/bundle.json","state":"https://pith.science/pith/RIUB5RACBJQL7EGEVUVT5YPBII/state.json","well_known_bundle":"https://pith.science/.well-known/pith/RIUB5RACBJQL7EGEVUVT5YPBII/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2023:RIUB5RACBJQL7EGEVUVT5YPBII","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"5119e51677a320cff9580c51b4f4191f723589b0d67b6f4286902afb3ae964e9","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-12-30T14:14:14Z","title_canon_sha256":"44baa034d026023c1e16ebb715568d6d2f8f0eda6145e1b5bb757b7fbc9664ba"},"schema_version":"1.0","source":{"id":"2401.00243","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2401.00243","created_at":"2026-07-05T07:29:02Z"},{"alias_kind":"arxiv_version","alias_value":"2401.00243v1","created_at":"2026-07-05T07:29:02Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.00243","created_at":"2026-07-05T07:29:02Z"},{"alias_kind":"pith_short_12","alias_value":"RIUB5RACBJQL","created_at":"2026-07-05T07:29:02Z"},{"alias_kind":"pith_short_16","alias_value":"RIUB5RACBJQL7EGE","created_at":"2026-07-05T07:29:02Z"},{"alias_kind":"pith_short_8","alias_value":"RIUB5RAC","created_at":"2026-07-05T07:29:02Z"}],"graph_snapshots":[{"event_id":"sha256:ef50553b775c541b34ed693deaeaf25dcc774c115e2246f651d9d9889f77fe39","target":"graph","created_at":"2026-07-05T07:29:02Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2401.00243/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement learning from human feedback (RLHF) emerges as a promising paradigm for aligning large language models (LLMs). However, a notable challenge in RLHF is overoptimization, where beyond a certain threshold, the pursuit of higher rewards leads to a decline in human preferences. In this paper, we observe the weakness of KL regularization which is commonly employed in existing RLHF methods to address overoptimization. To mitigate this limitation, we scrutinize the RLHF objective in the offline dataset and propose uncertainty-penalized RLHF (UP-RLHF), which incorporates uncertainty regul","authors_text":"Bo Ding, Dawei Feng, Han Zhang, Huaimin Wang, Kele Xu, Yuanzhao Zhai, Yue Yu, Yu Lei","cross_cats":[],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-12-30T14:14:14Z","title":"Uncertainty-Penalized Reinforcement Learning from Human Feedback with Diverse Reward LoRA Ensembles"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.00243","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:1a2dec4b4e8409c13707b2f8476d267cafe8125a56a70e8ef6891be5cb689aac","target":"record","created_at":"2026-07-05T07:29:02Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"5119e51677a320cff9580c51b4f4191f723589b0d67b6f4286902afb3ae964e9","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-12-30T14:14:14Z","title_canon_sha256":"44baa034d026023c1e16ebb715568d6d2f8f0eda6145e1b5bb757b7fbc9664ba"},"schema_version":"1.0","source":{"id":"2401.00243","kind":"arxiv","version":1}},"canonical_sha256":"8a281ec4020a60bf90c4ad2b3ee1e14237a643a9b70bcdc0e52a33af97e99329","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"8a281ec4020a60bf90c4ad2b3ee1e14237a643a9b70bcdc0e52a33af97e99329","first_computed_at":"2026-07-05T07:29:02.709690Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T07:29:02.709690Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"3S7zgy+zT/urUdEvh0y6Ems2EoEV600xtqTvbroh1O6FDkyIWgPR4uWBVr1EfKqsJbigHMPU79j1MVC1uB7+CQ==","signature_status":"signed_v1","signed_at":"2026-07-05T07:29:02.710299Z","signed_message":"canonical_sha256_bytes"},"source_id":"2401.00243","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:1a2dec4b4e8409c13707b2f8476d267cafe8125a56a70e8ef6891be5cb689aac","sha256:ef50553b775c541b34ed693deaeaf25dcc774c115e2246f651d9d9889f77fe39"],"state_sha256":"1f134db47386b4fbb41b92c83d2c42301893485bccc2f7a26d76b34d872d08a7"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"b2BST0JlVI3mCWqxG0MguaubG7Lif+xKTAOtBxCy4iHpTN6h1PhKNvQ/cNNa/YD+u1CzkUk6VzrWmLJX839BAw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-07-29T09:29:20.313529Z","bundle_sha256":"74a45d52b965cf22fb9e70ef3905ee988a1fe645c5b3344268ca1c6ae918157b"}}