{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:L4YR4RGUOMAGXKAOI3MN2KI2OS","short_pith_number":"pith:L4YR4RGU","schema_version":"1.0","canonical_sha256":"5f311e44d473006ba80e46d8dd291a7483f5f48ded40fcbb91e4e543934a54ff","source":{"kind":"arxiv","id":"2401.06080","version":2},"attestation_state":"computed","paper":{"title":"Secrets of RLHF in Large Language Models Part II: Reward Modeling","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Binghai Wang, Caishuang Huang, Chenyu Shi, Enyu Zhou, Hang Yan, Jun Zhao, Lixing Shen, Lu Chen, Nuo Xu, Qi Zhang, Rui Zheng, Senjie Jin, Shihan Dou, Songyang Gao, Tao Gui, Tao Ji, Wei Shen, Xiaoran Fan, Xiao Wang, Xipeng Qiu, Xuanjing Huang, Yan Liu, Yu-Gang Jiang, Yuhao Zhou, Zhan Chen, Zhiheng Xi, Zuxuan Wu","submitted_at":"2024-01-11T17:56:59Z","abstract_excerpt":"Reinforcement Learning from Human Feedback (RLHF) has become a crucial technology for aligning language models with human values and intentions, enabling models to produce more helpful and harmless responses. Reward models are trained as proxies for human preferences to drive reinforcement learning optimization. While reward models are often considered central to achieving high performance, they face the following challenges in practical applications: (1) Incorrect and ambiguous preference pairs in the dataset may hinder the reward model from accurately capturing human intent. (2) Reward model"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.06080","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2024-01-11T17:56:59Z","cross_cats_sorted":[],"title_canon_sha256":"9bc59c8055a435abbd703a9c8d769f052216263221491a5d8d240b1fbccc3390","abstract_canon_sha256":"b5cb2bb7a43f2ef38b68e830eb925e9c44251a95f4798f4a9cc8dafd7b30c922"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:32:45.902856Z","signature_b64":"/ThV2WbDhVSe7JfefQBIhtVSo9h/VU9oqMA8M4jo9E/CAcfof+1bczDwKuraLXtkWKvlm3BcYaXDlozKh2tCDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5f311e44d473006ba80e46d8dd291a7483f5f48ded40fcbb91e4e543934a54ff","last_reissued_at":"2026-07-05T07:32:45.902283Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:32:45.902283Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Secrets of RLHF in Large Language Models Part II: Reward Modeling","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Binghai Wang, Caishuang Huang, Chenyu Shi, Enyu Zhou, Hang Yan, Jun Zhao, Lixing Shen, Lu Chen, Nuo Xu, Qi Zhang, Rui Zheng, Senjie Jin, Shihan Dou, Songyang Gao, Tao Gui, Tao Ji, Wei Shen, Xiaoran Fan, Xiao Wang, Xipeng Qiu, Xuanjing Huang, Yan Liu, Yu-Gang Jiang, Yuhao Zhou, Zhan Chen, Zhiheng Xi, Zuxuan Wu","submitted_at":"2024-01-11T17:56:59Z","abstract_excerpt":"Reinforcement Learning from Human Feedback (RLHF) has become a crucial technology for aligning language models with human values and intentions, enabling models to produce more helpful and harmless responses. Reward models are trained as proxies for human preferences to drive reinforcement learning optimization. While reward models are often considered central to achieving high performance, they face the following challenges in practical applications: (1) Incorrect and ambiguous preference pairs in the dataset may hinder the reward model from accurately capturing human intent. (2) Reward model"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.06080","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.06080/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.06080","created_at":"2026-07-05T07:32:45.902356+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.06080v2","created_at":"2026-07-05T07:32:45.902356+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.06080","created_at":"2026-07-05T07:32:45.902356+00:00"},{"alias_kind":"pith_short_12","alias_value":"L4YR4RGUOMAG","created_at":"2026-07-05T07:32:45.902356+00:00"},{"alias_kind":"pith_short_16","alias_value":"L4YR4RGUOMAGXKAO","created_at":"2026-07-05T07:32:45.902356+00:00"},{"alias_kind":"pith_short_8","alias_value":"L4YR4RGU","created_at":"2026-07-05T07:32:45.902356+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":22,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21943","citing_title":"Modularized Reinforcement Learning on LLMs: From MDP Creation to Exploration and Learning","ref_index":210,"is_internal_anchor":false},{"citing_arxiv_id":"2606.13209","citing_title":"Understanding helpfulness and harmless tension in reward models","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10528","citing_title":"Representation-Aware Advantage Estimation: Your Reward Model Provides More Than A Scalar Output","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09711","citing_title":"Proxy Reward Internalization and Mechanistic Exploitation: A Learned Precursor to Reward Hacking and Its Generalization","ref_index":115,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28707","citing_title":"BV-Blend: Uncertainty-Weighted Historical Baselines for Stable Critic-Free RL with Verifiable Rewards","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05054","citing_title":"Boosting Self-Consistency with Ranking","ref_index":185,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09043","citing_title":"DynaCF: Mitigating Shortcut Learning in Reward Models via Dynamic Counterfactual Sensitivity","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2412.15115","citing_title":"Qwen2.5 Technical Report","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2502.06387","citing_title":"How Humans Help LLMs: Assessing and Incentivizing Human Preference Annotators","ref_index":88,"is_internal_anchor":false},{"citing_arxiv_id":"2505.19134","citing_title":"Incentivizing High-Quality Human Annotations with Golden Questions","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2510.13830","citing_title":"Users as Annotators: LLM Preference Learning from Comparison Mode","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2403.07691","citing_title":"ORPO: Monolithic Preference Optimization without Reference Model","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2411.10442","citing_title":"Enhancing the Reasoning Ability of Multimodal Large Language Models via Mixed Preference Optimization","ref_index":97,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15012","citing_title":"Boosting Reinforcement Learning with Verifiable Rewards via Randomly Selected Few-Shot Guidance","ref_index":79,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11974","citing_title":"Towards Order Fairness: Mitigating LLMs Order Sensitivity through Dual Group Advantage Optimization","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09119","citing_title":"Personalized Alignment Revisited: The Necessity and Sufficiency of User Diversity","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19544","citing_title":"DT2IT-MRM: Debiased Preference Construction and Iterative Training for Multimodal Reward Modeling","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11446","citing_title":"Low-rank Optimization Trajectories Modeling for LLM RLVR Acceleration","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2505.07062","citing_title":"Seed1.5-VL Technical Report","ref_index":138,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13602","citing_title":"Reward Hacking in the Era of Large Models: Mechanisms, Emergent Misalignment, Challenges","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14265","citing_title":"Reinforcement Learning via Value Gradient Flow","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02626","citing_title":"Gradient-Gated DPO: Stabilizing Preference Optimization in Language Models","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/L4YR4RGUOMAGXKAOI3MN2KI2OS","json":"https://pith.science/pith/L4YR4RGUOMAGXKAOI3MN2KI2OS.json","graph_json":"https://pith.science/api/pith-number/L4YR4RGUOMAGXKAOI3MN2KI2OS/graph.json","events_json":"https://pith.science/api/pith-number/L4YR4RGUOMAGXKAOI3MN2KI2OS/events.json","paper":"https://pith.science/paper/L4YR4RGU"},"agent_actions":{"view_html":"https://pith.science/pith/L4YR4RGUOMAGXKAOI3MN2KI2OS","download_json":"https://pith.science/pith/L4YR4RGUOMAGXKAOI3MN2KI2OS.json","view_paper":"https://pith.science/paper/L4YR4RGU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.06080&json=true","fetch_graph":"https://pith.science/api/pith-number/L4YR4RGUOMAGXKAOI3MN2KI2OS/graph.json","fetch_events":"https://pith.science/api/pith-number/L4YR4RGUOMAGXKAOI3MN2KI2OS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/L4YR4RGUOMAGXKAOI3MN2KI2OS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/L4YR4RGUOMAGXKAOI3MN2KI2OS/action/storage_attestation","attest_author":"https://pith.science/pith/L4YR4RGUOMAGXKAOI3MN2KI2OS/action/author_attestation","sign_citation":"https://pith.science/pith/L4YR4RGUOMAGXKAOI3MN2KI2OS/action/citation_signature","submit_replication":"https://pith.science/pith/L4YR4RGUOMAGXKAOI3MN2KI2OS/action/replication_record"}},"created_at":"2026-07-05T07:32:45.902356+00:00","updated_at":"2026-07-05T07:32:45.902356+00:00"}