{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:VY6CAMKNMILHNX3BKXVVUBGXTV","short_pith_number":"pith:VY6CAMKN","schema_version":"1.0","canonical_sha256":"ae3c20314d621676df6155eb5a04d79d48467c83cdec1e008d2f5cdf8baa5c8e","source":{"kind":"arxiv","id":"2505.12843","version":2},"attestation_state":"computed","paper":{"title":"Bias Fitting to Mitigate Length Bias of Reward Model in RLHF","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Dongyun Xue, Houqiang Li, Jianfeng Cai, Jinhua Zhu, Kangwen Zhao, Li Li, Ruopei Sun, Wengang Zhou","submitted_at":"2025-05-19T08:29:28Z","abstract_excerpt":"Reinforcement Learning from Human Feedback (RLHF) relies on reward models to align large language models with human preferences. However, RLHF often suffers from reward hacking, wherein policy learning exploits flaws in the trained reward model to maximize reward scores without genuinely aligning with human preferences. A significant example of such reward hacking is length bias, where reward models usually favor longer responses irrespective of actual response quality. Previous works on tackling length bias have notable limitations, these approaches either mitigate bias without characterizing"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.12843","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-19T08:29:28Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"f78ce95c7711845b3b0eb572e5e3ea1cfe6918640f1089524a73f54ed8266d27","abstract_canon_sha256":"ed09b76b56fc3504728515dea65f4cf3a1ef9c0bf901936362d2745e5dd6fe55"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-06-25T01:17:44.541200Z","signature_b64":"FaPuxXXIIyIP6iUrK+17SETR6/Uvasnz48QJUXOFOfmt4us7/8jT5LAPQ2sPtGwiyR2mGCAQA5TUjXhXSzR8DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ae3c20314d621676df6155eb5a04d79d48467c83cdec1e008d2f5cdf8baa5c8e","last_reissued_at":"2026-06-25T01:17:44.540713Z","signature_status":"signed_v1","first_computed_at":"2026-06-25T01:17:44.540713Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Bias Fitting to Mitigate Length Bias of Reward Model in RLHF","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Dongyun Xue, Houqiang Li, Jianfeng Cai, Jinhua Zhu, Kangwen Zhao, Li Li, Ruopei Sun, Wengang Zhou","submitted_at":"2025-05-19T08:29:28Z","abstract_excerpt":"Reinforcement Learning from Human Feedback (RLHF) relies on reward models to align large language models with human preferences. However, RLHF often suffers from reward hacking, wherein policy learning exploits flaws in the trained reward model to maximize reward scores without genuinely aligning with human preferences. A significant example of such reward hacking is length bias, where reward models usually favor longer responses irrespective of actual response quality. Previous works on tackling length bias have notable limitations, these approaches either mitigate bias without characterizing"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.12843","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.12843/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.12843","created_at":"2026-06-25T01:17:44.540771+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.12843v2","created_at":"2026-06-25T01:17:44.540771+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.12843","created_at":"2026-06-25T01:17:44.540771+00:00"},{"alias_kind":"pith_short_12","alias_value":"VY6CAMKNMILH","created_at":"2026-06-25T01:17:44.540771+00:00"},{"alias_kind":"pith_short_16","alias_value":"VY6CAMKNMILHNX3B","created_at":"2026-06-25T01:17:44.540771+00:00"},{"alias_kind":"pith_short_8","alias_value":"VY6CAMKN","created_at":"2026-06-25T01:17:44.540771+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":3,"sample":[{"citing_arxiv_id":"2605.27996","citing_title":"Reward Bias Substitution: Single-Axis Bias Mitigations Redirect Optimization Pressure","ref_index":96,"is_internal_anchor":true},{"citing_arxiv_id":"2604.13602","citing_title":"Reward Hacking in the Era of Large Models: Mechanisms, Emergent Misalignment, Challenges","ref_index":130,"is_internal_anchor":true},{"citing_arxiv_id":"2604.17328","citing_title":"Rethinking the Comparison Unit in Sequence-Level Reinforcement Learning: An Equal-Length Paired Training Framework from Loss Correction to Sample Construction","ref_index":15,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VY6CAMKNMILHNX3BKXVVUBGXTV","json":"https://pith.science/pith/VY6CAMKNMILHNX3BKXVVUBGXTV.json","graph_json":"https://pith.science/api/pith-number/VY6CAMKNMILHNX3BKXVVUBGXTV/graph.json","events_json":"https://pith.science/api/pith-number/VY6CAMKNMILHNX3BKXVVUBGXTV/events.json","paper":"https://pith.science/paper/VY6CAMKN"},"agent_actions":{"view_html":"https://pith.science/pith/VY6CAMKNMILHNX3BKXVVUBGXTV","download_json":"https://pith.science/pith/VY6CAMKNMILHNX3BKXVVUBGXTV.json","view_paper":"https://pith.science/paper/VY6CAMKN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.12843&json=true","fetch_graph":"https://pith.science/api/pith-number/VY6CAMKNMILHNX3BKXVVUBGXTV/graph.json","fetch_events":"https://pith.science/api/pith-number/VY6CAMKNMILHNX3BKXVVUBGXTV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VY6CAMKNMILHNX3BKXVVUBGXTV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VY6CAMKNMILHNX3BKXVVUBGXTV/action/storage_attestation","attest_author":"https://pith.science/pith/VY6CAMKNMILHNX3BKXVVUBGXTV/action/author_attestation","sign_citation":"https://pith.science/pith/VY6CAMKNMILHNX3BKXVVUBGXTV/action/citation_signature","submit_replication":"https://pith.science/pith/VY6CAMKNMILHNX3BKXVVUBGXTV/action/replication_record"}},"created_at":"2026-06-25T01:17:44.540771+00:00","updated_at":"2026-06-25T01:17:44.540771+00:00"}