{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ASN56S5CBHTSR7MCMB5X7OPDQL","short_pith_number":"pith:ASN56S5C","schema_version":"1.0","canonical_sha256":"049bdf4ba209e728fd82607b7fb9e382e301681bef2cf3ffe1b0a86c1e40e009","source":{"kind":"arxiv","id":"2406.10216","version":2},"attestation_state":"computed","paper":{"title":"Regularizing Hidden States Enables Learning Generalizable Reward Model for LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Huan Zhang, Rui Yang, Ruomeng Ding, Tong Zhang, Yong Lin","submitted_at":"2024-06-14T17:49:59Z","abstract_excerpt":"Reward models trained on human preference data have been proven to effectively align Large Language Models (LLMs) with human intent within the framework of reinforcement learning from human feedback (RLHF). However, current reward models have limited generalization capabilities to unseen prompts and responses, which can lead to an unexpected phenomenon known as reward over-optimization, resulting in a decline in actual performance due to excessive optimization of rewards. While previous research has advocated for constraining policy optimization, our study introduces a novel approach to enhanc"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.10216","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-06-14T17:49:59Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"cb7d709f77324a677b0269d5af199b4e948ea13c15ebad5666676a21ad616ea0","abstract_canon_sha256":"62db7e74d90b88a4425c3ce4b9d4df76b62e0418111886c859bbfb35cbaa4670"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:24:36.088004Z","signature_b64":"7RcHtdhMWn4Sq/Jww2N4KrhiCPpJZlq29Q9sSPlHdIA0fKuG5HD6JXI1brxsFYxEQiCQTJLrftN/Yr/iKC22Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"049bdf4ba209e728fd82607b7fb9e382e301681bef2cf3ffe1b0a86c1e40e009","last_reissued_at":"2026-07-05T09:24:36.087497Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:24:36.087497Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Regularizing Hidden States Enables Learning Generalizable Reward Model for LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Huan Zhang, Rui Yang, Ruomeng Ding, Tong Zhang, Yong Lin","submitted_at":"2024-06-14T17:49:59Z","abstract_excerpt":"Reward models trained on human preference data have been proven to effectively align Large Language Models (LLMs) with human intent within the framework of reinforcement learning from human feedback (RLHF). However, current reward models have limited generalization capabilities to unseen prompts and responses, which can lead to an unexpected phenomenon known as reward over-optimization, resulting in a decline in actual performance due to excessive optimization of rewards. While previous research has advocated for constraining policy optimization, our study introduces a novel approach to enhanc"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.10216","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.10216/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.10216","created_at":"2026-07-05T09:24:36.087557+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.10216v2","created_at":"2026-07-05T09:24:36.087557+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.10216","created_at":"2026-07-05T09:24:36.087557+00:00"},{"alias_kind":"pith_short_12","alias_value":"ASN56S5CBHTS","created_at":"2026-07-05T09:24:36.087557+00:00"},{"alias_kind":"pith_short_16","alias_value":"ASN56S5CBHTSR7MC","created_at":"2026-07-05T09:24:36.087557+00:00"},{"alias_kind":"pith_short_8","alias_value":"ASN56S5C","created_at":"2026-07-05T09:24:36.087557+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03131","citing_title":"HARVE: Hacking-Aware Reward-Head Vector Editing for Robust Reward Models","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31748","citing_title":"Addressing Over-Refusal in LLMs with Competing Rewards","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09043","citing_title":"DynaCF: Mitigating Shortcut Learning in Reward Models via Dynamic Counterfactual Sensitivity","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2412.08812","citing_title":"Test-Time Alignment via Hypothesis Reweighting","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2410.18451","citing_title":"Skywork-Reward: Bag of Tricks for Reward Modeling in LLMs","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ASN56S5CBHTSR7MCMB5X7OPDQL","json":"https://pith.science/pith/ASN56S5CBHTSR7MCMB5X7OPDQL.json","graph_json":"https://pith.science/api/pith-number/ASN56S5CBHTSR7MCMB5X7OPDQL/graph.json","events_json":"https://pith.science/api/pith-number/ASN56S5CBHTSR7MCMB5X7OPDQL/events.json","paper":"https://pith.science/paper/ASN56S5C"},"agent_actions":{"view_html":"https://pith.science/pith/ASN56S5CBHTSR7MCMB5X7OPDQL","download_json":"https://pith.science/pith/ASN56S5CBHTSR7MCMB5X7OPDQL.json","view_paper":"https://pith.science/paper/ASN56S5C","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.10216&json=true","fetch_graph":"https://pith.science/api/pith-number/ASN56S5CBHTSR7MCMB5X7OPDQL/graph.json","fetch_events":"https://pith.science/api/pith-number/ASN56S5CBHTSR7MCMB5X7OPDQL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ASN56S5CBHTSR7MCMB5X7OPDQL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ASN56S5CBHTSR7MCMB5X7OPDQL/action/storage_attestation","attest_author":"https://pith.science/pith/ASN56S5CBHTSR7MCMB5X7OPDQL/action/author_attestation","sign_citation":"https://pith.science/pith/ASN56S5CBHTSR7MCMB5X7OPDQL/action/citation_signature","submit_replication":"https://pith.science/pith/ASN56S5CBHTSR7MCMB5X7OPDQL/action/replication_record"}},"created_at":"2026-07-05T09:24:36.087557+00:00","updated_at":"2026-07-05T09:24:36.087557+00:00"}