{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:DZNJNSIGD6NQ4IVFKAVQMITQTJ","short_pith_number":"pith:DZNJNSIG","schema_version":"1.0","canonical_sha256":"1e5a96c9061f9b0e22a5502b0622709a508d374cc9730ed1dbb0f877f9a29a6b","source":{"kind":"arxiv","id":"2501.09620","version":2},"attestation_state":"computed","paper":{"title":"Beyond Reward Hacking: Causal Rewards for Large Language Model Alignment","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Chaoqi Wang, Chen Zhu, Hao Ma, Jiayi Liu, Lizhu Zhang, Sinong Wang, Xiangjun Fan, Yibo Jiang, Yuxin Chen, Zhaorun Chen, Zhuokai Zhao","submitted_at":"2025-01-16T16:00:37Z","abstract_excerpt":"Recent advances in large language models (LLMs) have demonstrated significant progress in performing complex tasks. While Reinforcement Learning from Human Feedback (RLHF) has been effective in aligning LLMs with human preferences, it is susceptible to spurious correlations in reward modeling. Consequently, it often introduces biases-such as length bias, sycophancy, conceptual bias, and discrimination-that hinder the model's ability to capture true causal relationships. To address this, we propose a novel causal reward modeling approach that integrates causality to mitigate these spurious corr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.09620","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-01-16T16:00:37Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"7e7d276700063c9a8be1681c6bb748fb753ba5942c0875e812154fc75209cbf7","abstract_canon_sha256":"31bdabe9602a1ec294beecf7d7f100007d03bcd3810cd95bad480ace2abc4152"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:11:39.807555Z","signature_b64":"8Hcc4ABLD3+AsuJWIIOQDNrqYkr1+jUvC4HmhYCoKes0/PzyaZzW+jd1TiBJTyHCLXM2mnSkmqRkcal7K5BkCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1e5a96c9061f9b0e22a5502b0622709a508d374cc9730ed1dbb0f877f9a29a6b","last_reissued_at":"2026-07-05T11:11:39.807014Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:11:39.807014Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Beyond Reward Hacking: Causal Rewards for Large Language Model Alignment","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Chaoqi Wang, Chen Zhu, Hao Ma, Jiayi Liu, Lizhu Zhang, Sinong Wang, Xiangjun Fan, Yibo Jiang, Yuxin Chen, Zhaorun Chen, Zhuokai Zhao","submitted_at":"2025-01-16T16:00:37Z","abstract_excerpt":"Recent advances in large language models (LLMs) have demonstrated significant progress in performing complex tasks. While Reinforcement Learning from Human Feedback (RLHF) has been effective in aligning LLMs with human preferences, it is susceptible to spurious correlations in reward modeling. Consequently, it often introduces biases-such as length bias, sycophancy, conceptual bias, and discrimination-that hinder the model's ability to capture true causal relationships. To address this, we propose a novel causal reward modeling approach that integrates causality to mitigate these spurious corr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.09620","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.09620/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.09620","created_at":"2026-07-05T11:11:39.807083+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.09620v2","created_at":"2026-07-05T11:11:39.807083+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.09620","created_at":"2026-07-05T11:11:39.807083+00:00"},{"alias_kind":"pith_short_12","alias_value":"DZNJNSIGD6NQ","created_at":"2026-07-05T11:11:39.807083+00:00"},{"alias_kind":"pith_short_16","alias_value":"DZNJNSIGD6NQ4IVF","created_at":"2026-07-05T11:11:39.807083+00:00"},{"alias_kind":"pith_short_8","alias_value":"DZNJNSIG","created_at":"2026-07-05T11:11:39.807083+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09711","citing_title":"Proxy Reward Internalization and Mechanistic Exploitation: A Learned Precursor to Reward Hacking and Its Generalization","ref_index":135,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24076","citing_title":"Causality as the Statistical Conscience of Artificial Intelligence: From Pearl's Ladder to Trustworthy Machines","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27996","citing_title":"Reward Bias Substitution: Single-Axis Bias Mitigations Redirect Optimization Pressure","ref_index":85,"is_internal_anchor":false},{"citing_arxiv_id":"2601.05106","citing_title":"Token-Level LLM Collaboration via FusionRoute","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18721","citing_title":"General Preference Reinforcement Learning","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2601.21350","citing_title":"Factored Causal Representation Learning for Robust Reward Modeling in RLHF","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18721","citing_title":"General Preference Reinforcement Learning","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16339","citing_title":"Preference Instability in Reward Models: Detection and Mitigation via Sparse Autoencoders","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18721","citing_title":"General Preference Reinforcement Learning","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2602.06475","citing_title":"Towards Generalizable Reasoning: Group Causal Counterfactual Policy Optimization for LLM Reasoning","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13833","citing_title":"Robust Reward Modeling for Large Language Models via Causal Decomposition","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DZNJNSIGD6NQ4IVFKAVQMITQTJ","json":"https://pith.science/pith/DZNJNSIGD6NQ4IVFKAVQMITQTJ.json","graph_json":"https://pith.science/api/pith-number/DZNJNSIGD6NQ4IVFKAVQMITQTJ/graph.json","events_json":"https://pith.science/api/pith-number/DZNJNSIGD6NQ4IVFKAVQMITQTJ/events.json","paper":"https://pith.science/paper/DZNJNSIG"},"agent_actions":{"view_html":"https://pith.science/pith/DZNJNSIGD6NQ4IVFKAVQMITQTJ","download_json":"https://pith.science/pith/DZNJNSIGD6NQ4IVFKAVQMITQTJ.json","view_paper":"https://pith.science/paper/DZNJNSIG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.09620&json=true","fetch_graph":"https://pith.science/api/pith-number/DZNJNSIGD6NQ4IVFKAVQMITQTJ/graph.json","fetch_events":"https://pith.science/api/pith-number/DZNJNSIGD6NQ4IVFKAVQMITQTJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DZNJNSIGD6NQ4IVFKAVQMITQTJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DZNJNSIGD6NQ4IVFKAVQMITQTJ/action/storage_attestation","attest_author":"https://pith.science/pith/DZNJNSIGD6NQ4IVFKAVQMITQTJ/action/author_attestation","sign_citation":"https://pith.science/pith/DZNJNSIGD6NQ4IVFKAVQMITQTJ/action/citation_signature","submit_replication":"https://pith.science/pith/DZNJNSIGD6NQ4IVFKAVQMITQTJ/action/replication_record"}},"created_at":"2026-07-05T11:11:39.807083+00:00","updated_at":"2026-07-05T11:11:39.807083+00:00"}