{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:VY6CAMKNMILHNX3BKXVVUBGXTV","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"ed09b76b56fc3504728515dea65f4cf3a1ef9c0bf901936362d2745e5dd6fe55","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-19T08:29:28Z","title_canon_sha256":"f78ce95c7711845b3b0eb572e5e3ea1cfe6918640f1089524a73f54ed8266d27"},"schema_version":"1.0","source":{"id":"2505.12843","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2505.12843","created_at":"2026-06-25T01:17:44Z"},{"alias_kind":"arxiv_version","alias_value":"2505.12843v2","created_at":"2026-06-25T01:17:44Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.12843","created_at":"2026-06-25T01:17:44Z"},{"alias_kind":"pith_short_12","alias_value":"VY6CAMKNMILH","created_at":"2026-06-25T01:17:44Z"},{"alias_kind":"pith_short_16","alias_value":"VY6CAMKNMILHNX3B","created_at":"2026-06-25T01:17:44Z"},{"alias_kind":"pith_short_8","alias_value":"VY6CAMKN","created_at":"2026-06-25T01:17:44Z"}],"graph_snapshots":[{"event_id":"sha256:a6b980fd8b9610bb6b532f41958d99246b1143f5c6c16ca6492a2528e9ea80dd","target":"graph","created_at":"2026-06-25T01:17:44Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2505.12843/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement Learning from Human Feedback (RLHF) relies on reward models to align large language models with human preferences. However, RLHF often suffers from reward hacking, wherein policy learning exploits flaws in the trained reward model to maximize reward scores without genuinely aligning with human preferences. A significant example of such reward hacking is length bias, where reward models usually favor longer responses irrespective of actual response quality. Previous works on tackling length bias have notable limitations, these approaches either mitigate bias without characterizing","authors_text":"Dongyun Xue, Houqiang Li, Jianfeng Cai, Jinhua Zhu, Kangwen Zhao, Li Li, Ruopei Sun, Wengang Zhou","cross_cats":["cs.AI"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-19T08:29:28Z","title":"Bias Fitting to Mitigate Length Bias of Reward Model in RLHF"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.12843","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:2ce6750bbc528c700f14735c32b5147c3d81f69d7e886dd4bf25738e3a5384c8","target":"record","created_at":"2026-06-25T01:17:44Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"ed09b76b56fc3504728515dea65f4cf3a1ef9c0bf901936362d2745e5dd6fe55","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-19T08:29:28Z","title_canon_sha256":"f78ce95c7711845b3b0eb572e5e3ea1cfe6918640f1089524a73f54ed8266d27"},"schema_version":"1.0","source":{"id":"2505.12843","kind":"arxiv","version":2}},"canonical_sha256":"ae3c20314d621676df6155eb5a04d79d48467c83cdec1e008d2f5cdf8baa5c8e","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"ae3c20314d621676df6155eb5a04d79d48467c83cdec1e008d2f5cdf8baa5c8e","first_computed_at":"2026-06-25T01:17:44.540713Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-06-25T01:17:44.540713Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"FaPuxXXIIyIP6iUrK+17SETR6/Uvasnz48QJUXOFOfmt4us7/8jT5LAPQ2sPtGwiyR2mGCAQA5TUjXhXSzR8DQ==","signature_status":"signed_v1","signed_at":"2026-06-25T01:17:44.541200Z","signed_message":"canonical_sha256_bytes"},"source_id":"2505.12843","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:2ce6750bbc528c700f14735c32b5147c3d81f69d7e886dd4bf25738e3a5384c8","sha256:a6b980fd8b9610bb6b532f41958d99246b1143f5c6c16ca6492a2528e9ea80dd"],"state_sha256":"d21dfa2338b6c6ad92d12d28c1557017ba72487c72d115a6cf76febc05cb0f4f"}