{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:KN4XLYFFHF2Q5R3ERYDWPLIKGT","short_pith_number":"pith:KN4XLYFF","schema_version":"1.0","canonical_sha256":"537975e0a539750ec7648e0767ad0a34d0c3f69bfd438ddca04d97e2f14d7b2c","source":{"kind":"arxiv","id":"2409.15360","version":3},"attestation_state":"computed","paper":{"title":"Reward-Robust RLHF in LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Chao Yu, Dong Yan, Jialian Li, Jian Xie, Xingzhou Lou, Yiping Zhang, Yuan Shen, Yu Wang, Yuzi Yan","submitted_at":"2024-09-18T02:35:41Z","abstract_excerpt":"As Large Language Models (LLMs) continue to progress toward more advanced forms of intelligence, Reinforcement Learning from Human Feedback (RLHF) is increasingly seen as a key pathway toward achieving Artificial General Intelligence (AGI). However, the reliance on reward-model-based (RM-based) alignment methods introduces significant challenges due to the inherent instability and imperfections of Reward Models (RMs), which can lead to critical issues such as reward hacking and misalignment with human intentions. In this paper, we introduce a reward-robust RLHF framework aimed at addressing th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.15360","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-09-18T02:35:41Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"1080627b3081518dbe421875b24ddb43f46812acdf80da61149b8039a786ad59","abstract_canon_sha256":"9308465f6c3f21be2d81cbd5bac84d94569a3ff21979e7a3c9b8d1d688fd4d79"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:21:28.367839Z","signature_b64":"G/MeBI0Py07XyGW+uH2wKMtN7koPl1eDF/A5nL+suAchzgfB6pxK4w3zpkwfv4jiyIYxVgFJMAoArDrQjOjsBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"537975e0a539750ec7648e0767ad0a34d0c3f69bfd438ddca04d97e2f14d7b2c","last_reissued_at":"2026-07-05T09:21:28.367341Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:21:28.367341Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Reward-Robust RLHF in LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Chao Yu, Dong Yan, Jialian Li, Jian Xie, Xingzhou Lou, Yiping Zhang, Yuan Shen, Yu Wang, Yuzi Yan","submitted_at":"2024-09-18T02:35:41Z","abstract_excerpt":"As Large Language Models (LLMs) continue to progress toward more advanced forms of intelligence, Reinforcement Learning from Human Feedback (RLHF) is increasingly seen as a key pathway toward achieving Artificial General Intelligence (AGI). However, the reliance on reward-model-based (RM-based) alignment methods introduces significant challenges due to the inherent instability and imperfections of Reward Models (RMs), which can lead to critical issues such as reward hacking and misalignment with human intentions. In this paper, we introduce a reward-robust RLHF framework aimed at addressing th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.15360","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.15360/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.15360","created_at":"2026-07-05T09:21:28.367408+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.15360v3","created_at":"2026-07-05T09:21:28.367408+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.15360","created_at":"2026-07-05T09:21:28.367408+00:00"},{"alias_kind":"pith_short_12","alias_value":"KN4XLYFFHF2Q","created_at":"2026-07-05T09:21:28.367408+00:00"},{"alias_kind":"pith_short_16","alias_value":"KN4XLYFFHF2Q5R3E","created_at":"2026-07-05T09:21:28.367408+00:00"},{"alias_kind":"pith_short_8","alias_value":"KN4XLYFF","created_at":"2026-07-05T09:21:28.367408+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09711","citing_title":"Proxy Reward Internalization and Mechanistic Exploitation: A Learned Precursor to Reward Hacking and Its Generalization","ref_index":233,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09078","citing_title":"The Hidden Bias of Process Reward Models:PRISM for Rewarding the Right Reasoning","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09073","citing_title":"A Unifying Lens on Reward Uncertainty in RLHF","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02495","citing_title":"Efficient Preference Poisoning Attack on Offline RLHF","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25252","citing_title":"Quantifying Empirical Compute-Supervision Tradeoffs in RLVR","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01123","citing_title":"PERSA: Reinforcement Learning for Professor-Style Personalized Feedback with LLMs","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00155","citing_title":"Wasserstein Distributionally Robust Regret Optimization for Reinforcement Learning from Human Feedback","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13602","citing_title":"Reward Hacking in the Era of Large Models: Mechanisms, Emergent Misalignment, Challenges","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17197","citing_title":"Learning to Control Summaries with Score Ranking","ref_index":46,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KN4XLYFFHF2Q5R3ERYDWPLIKGT","json":"https://pith.science/pith/KN4XLYFFHF2Q5R3ERYDWPLIKGT.json","graph_json":"https://pith.science/api/pith-number/KN4XLYFFHF2Q5R3ERYDWPLIKGT/graph.json","events_json":"https://pith.science/api/pith-number/KN4XLYFFHF2Q5R3ERYDWPLIKGT/events.json","paper":"https://pith.science/paper/KN4XLYFF"},"agent_actions":{"view_html":"https://pith.science/pith/KN4XLYFFHF2Q5R3ERYDWPLIKGT","download_json":"https://pith.science/pith/KN4XLYFFHF2Q5R3ERYDWPLIKGT.json","view_paper":"https://pith.science/paper/KN4XLYFF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.15360&json=true","fetch_graph":"https://pith.science/api/pith-number/KN4XLYFFHF2Q5R3ERYDWPLIKGT/graph.json","fetch_events":"https://pith.science/api/pith-number/KN4XLYFFHF2Q5R3ERYDWPLIKGT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KN4XLYFFHF2Q5R3ERYDWPLIKGT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KN4XLYFFHF2Q5R3ERYDWPLIKGT/action/storage_attestation","attest_author":"https://pith.science/pith/KN4XLYFFHF2Q5R3ERYDWPLIKGT/action/author_attestation","sign_citation":"https://pith.science/pith/KN4XLYFFHF2Q5R3ERYDWPLIKGT/action/citation_signature","submit_replication":"https://pith.science/pith/KN4XLYFFHF2Q5R3ERYDWPLIKGT/action/replication_record"}},"created_at":"2026-07-05T09:21:28.367408+00:00","updated_at":"2026-07-05T09:21:28.367408+00:00"}