{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:W67GST2ZNKQZ3MTAQCT6K7YAOH","short_pith_number":"pith:W67GST2Z","schema_version":"1.0","canonical_sha256":"b7be694f596aa19db26080a7e57f0071fd78eb526f89023c1d7e81a4f5a9fda8","source":{"kind":"arxiv","id":"2502.18770","version":6},"attestation_state":"computed","paper":{"title":"Reward Shaping to Mitigate Reward Hacking in RLHF","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Chengyuan Yao, Heng Wang, Jiayi Fu, Qi Han, Xuandong Zhao, Yanghua Xiao","submitted_at":"2025-02-26T02:57:59Z","abstract_excerpt":"Reinforcement learning from human feedback (RLHF) is widely used to align large language models (LLMs) with human preferences. However, RLHF remains vulnerable to \\emph{reward hacking}, whereby a policy exploits imperfections in the reward function instead of learning the intended behavior, thereby undermining alignment. Although reward shaping can stabilize RLHF training and partially mitigate reward hacking, shaping methods and their underlying design principles have not been systematically investigated. To address this gap, we conduct a comprehensive study of prevalent reward-shaping techni"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.18770","kind":"arxiv","version":6},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-26T02:57:59Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"15b3dec513e766d045e101841c86589e5f48ddfc14d9c0a1ce67e361334a0243","abstract_canon_sha256":"ce461e7c445c3427ab55b16ceaaf18b50460ffe0062e3041f1cc8e7325f675b3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-07T00:50:51.707354Z","signature_b64":"CIjF+8ODD9Bjao15khuM9cudHiQbz0++bSWncHI2hiqcv9ok1TcZpcnz4cXdsCQc6Du1EQ/Rd/cSmthsgaB3Dw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b7be694f596aa19db26080a7e57f0071fd78eb526f89023c1d7e81a4f5a9fda8","last_reissued_at":"2026-08-07T00:50:51.705601Z","signature_status":"signed_v1","first_computed_at":"2026-08-07T00:50:51.705601Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Reward Shaping to Mitigate Reward Hacking in RLHF","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Chengyuan Yao, Heng Wang, Jiayi Fu, Qi Han, Xuandong Zhao, Yanghua Xiao","submitted_at":"2025-02-26T02:57:59Z","abstract_excerpt":"Reinforcement learning from human feedback (RLHF) is widely used to align large language models (LLMs) with human preferences. However, RLHF remains vulnerable to \\emph{reward hacking}, whereby a policy exploits imperfections in the reward function instead of learning the intended behavior, thereby undermining alignment. Although reward shaping can stabilize RLHF training and partially mitigate reward hacking, shaping methods and their underlying design principles have not been systematically investigated. To address this gap, we conduct a comprehensive study of prevalent reward-shaping techni"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.18770","kind":"arxiv","version":6},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.18770/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.18770","created_at":"2026-08-07T00:50:51.707198+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.18770v6","created_at":"2026-08-07T00:50:51.707198+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.18770","created_at":"2026-08-07T00:50:51.707198+00:00"},{"alias_kind":"pith_short_12","alias_value":"W67GST2ZNKQZ","created_at":"2026-08-07T00:50:51.707198+00:00"},{"alias_kind":"pith_short_16","alias_value":"W67GST2ZNKQZ3MTA","created_at":"2026-08-07T00:50:51.707198+00:00"},{"alias_kind":"pith_short_8","alias_value":"W67GST2Z","created_at":"2026-08-07T00:50:51.707198+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":20,"internal_anchor_count":20,"sample":[{"citing_arxiv_id":"2606.25757","citing_title":"OPERA: Aligning Open-Ended Reasoning via Objective Perplexity-based Reinforcement Learning","ref_index":1,"is_internal_anchor":true},{"citing_arxiv_id":"2606.19818","citing_title":"Uncertainty-Aware Reward Modeling for Stable RLHF","ref_index":8,"is_internal_anchor":true},{"citing_arxiv_id":"2606.10528","citing_title":"Representation-Aware Advantage Estimation: Your Reward Model Provides More Than A Scalar Output","ref_index":32,"is_internal_anchor":true},{"citing_arxiv_id":"2606.11052","citing_title":"Attention Amnesia in Hybrid LLMs: When CoT Fine-Tuning Breaks Long-Range Recall, and How to Fix It","ref_index":18,"is_internal_anchor":true},{"citing_arxiv_id":"2606.17682","citing_title":"From Trainee to Trainer: LLM-Designed Training Environment for RL with Multi-Agent Reasoning","ref_index":18,"is_internal_anchor":true},{"citing_arxiv_id":"2502.13957","citing_title":"Supervising the search process produces reliable and generalizable information-seeking agents","ref_index":16,"is_internal_anchor":true},{"citing_arxiv_id":"2605.18721","citing_title":"General Preference Reinforcement Learning","ref_index":21,"is_internal_anchor":true},{"citing_arxiv_id":"2601.21350","citing_title":"Factored Causal Representation Learning for Robust Reward Modeling in RLHF","ref_index":11,"is_internal_anchor":true},{"citing_arxiv_id":"2605.18721","citing_title":"General Preference Reinforcement Learning","ref_index":21,"is_internal_anchor":true},{"citing_arxiv_id":"2605.18721","citing_title":"General Preference Reinforcement Learning","ref_index":21,"is_internal_anchor":true},{"citing_arxiv_id":"2508.19652","citing_title":"Self-Rewarding Vision-Language Model via Reasoning Decomposition","ref_index":7,"is_internal_anchor":true},{"citing_arxiv_id":"2605.14220","citing_title":"Diagnosing Training Inference Mismatch in LLM Reinforcement Learning","ref_index":51,"is_internal_anchor":true},{"citing_arxiv_id":"2605.11865","citing_title":"Variance-aware Reward Modeling with Anchor Guidance","ref_index":49,"is_internal_anchor":true},{"citing_arxiv_id":"2605.12474","citing_title":"Reward Hacking in Rubric-Based Reinforcement Learning","ref_index":9,"is_internal_anchor":true},{"citing_arxiv_id":"2502.17419","citing_title":"From System 1 to System 2: A Survey of Reasoning Large Language Models","ref_index":170,"is_internal_anchor":true},{"citing_arxiv_id":"2605.08496","citing_title":"Latent Personality Alignment: Improving Harmlessness Without Mentioning Harms","ref_index":5,"is_internal_anchor":true},{"citing_arxiv_id":"2605.06036","citing_title":"Optimal Transport for LLM Reward Modeling from Noisy Preference","ref_index":260,"is_internal_anchor":true},{"citing_arxiv_id":"2604.21268","citing_title":"Measure Twice, Click Once: Co-evolving Proposer and Visual Critic via Reinforcement Learning for GUI Grounding","ref_index":31,"is_internal_anchor":true},{"citing_arxiv_id":"2604.17328","citing_title":"Rethinking the Comparison Unit in Sequence-Level Reinforcement Learning: An Equal-Length Paired Training Framework from Loss Correction to Sample Construction","ref_index":18,"is_internal_anchor":true},{"citing_arxiv_id":"2605.04431","citing_title":"Towards Robust LLM Post-Training: Automatic Failure Management for Reinforcement Fine-Tuning","ref_index":10,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/W67GST2ZNKQZ3MTAQCT6K7YAOH","json":"https://pith.science/pith/W67GST2ZNKQZ3MTAQCT6K7YAOH.json","graph_json":"https://pith.science/api/pith-number/W67GST2ZNKQZ3MTAQCT6K7YAOH/graph.json","events_json":"https://pith.science/api/pith-number/W67GST2ZNKQZ3MTAQCT6K7YAOH/events.json","paper":"https://pith.science/paper/W67GST2Z"},"agent_actions":{"view_html":"https://pith.science/pith/W67GST2ZNKQZ3MTAQCT6K7YAOH","download_json":"https://pith.science/pith/W67GST2ZNKQZ3MTAQCT6K7YAOH.json","view_paper":"https://pith.science/paper/W67GST2Z","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.18770&json=true","fetch_graph":"https://pith.science/api/pith-number/W67GST2ZNKQZ3MTAQCT6K7YAOH/graph.json","fetch_events":"https://pith.science/api/pith-number/W67GST2ZNKQZ3MTAQCT6K7YAOH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/W67GST2ZNKQZ3MTAQCT6K7YAOH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/W67GST2ZNKQZ3MTAQCT6K7YAOH/action/storage_attestation","attest_author":"https://pith.science/pith/W67GST2ZNKQZ3MTAQCT6K7YAOH/action/author_attestation","sign_citation":"https://pith.science/pith/W67GST2ZNKQZ3MTAQCT6K7YAOH/action/citation_signature","submit_replication":"https://pith.science/pith/W67GST2ZNKQZ3MTAQCT6K7YAOH/action/replication_record"}},"created_at":"2026-08-07T00:50:51.707198+00:00","updated_at":"2026-08-07T00:50:51.707198+00:00"}