{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:KN4XLYFFHF2Q5R3ERYDWPLIKGT","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"9308465f6c3f21be2d81cbd5bac84d94569a3ff21979e7a3c9b8d1d688fd4d79","cross_cats_sorted":["cs.AI","cs.CL"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-09-18T02:35:41Z","title_canon_sha256":"1080627b3081518dbe421875b24ddb43f46812acdf80da61149b8039a786ad59"},"schema_version":"1.0","source":{"id":"2409.15360","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2409.15360","created_at":"2026-07-05T09:21:28Z"},{"alias_kind":"arxiv_version","alias_value":"2409.15360v3","created_at":"2026-07-05T09:21:28Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.15360","created_at":"2026-07-05T09:21:28Z"},{"alias_kind":"pith_short_12","alias_value":"KN4XLYFFHF2Q","created_at":"2026-07-05T09:21:28Z"},{"alias_kind":"pith_short_16","alias_value":"KN4XLYFFHF2Q5R3E","created_at":"2026-07-05T09:21:28Z"},{"alias_kind":"pith_short_8","alias_value":"KN4XLYFF","created_at":"2026-07-05T09:21:28Z"}],"graph_snapshots":[{"event_id":"sha256:555c292cf2494faab9f2c3f207acb1e414782e4118ad2ab1e25a948fb1ae605a","target":"graph","created_at":"2026-07-05T09:21:28Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2409.15360/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"As Large Language Models (LLMs) continue to progress toward more advanced forms of intelligence, Reinforcement Learning from Human Feedback (RLHF) is increasingly seen as a key pathway toward achieving Artificial General Intelligence (AGI). However, the reliance on reward-model-based (RM-based) alignment methods introduces significant challenges due to the inherent instability and imperfections of Reward Models (RMs), which can lead to critical issues such as reward hacking and misalignment with human intentions. In this paper, we introduce a reward-robust RLHF framework aimed at addressing th","authors_text":"Chao Yu, Dong Yan, Jialian Li, Jian Xie, Xingzhou Lou, Yiping Zhang, Yuan Shen, Yu Wang, Yuzi Yan","cross_cats":["cs.AI","cs.CL"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-09-18T02:35:41Z","title":"Reward-Robust RLHF in LLMs"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.15360","kind":"arxiv","version":3},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:7d376341a73a2aae0e36b3297f78944a9312b60c189e15bd137393b34a9e6368","target":"record","created_at":"2026-07-05T09:21:28Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"9308465f6c3f21be2d81cbd5bac84d94569a3ff21979e7a3c9b8d1d688fd4d79","cross_cats_sorted":["cs.AI","cs.CL"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-09-18T02:35:41Z","title_canon_sha256":"1080627b3081518dbe421875b24ddb43f46812acdf80da61149b8039a786ad59"},"schema_version":"1.0","source":{"id":"2409.15360","kind":"arxiv","version":3}},"canonical_sha256":"537975e0a539750ec7648e0767ad0a34d0c3f69bfd438ddca04d97e2f14d7b2c","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"537975e0a539750ec7648e0767ad0a34d0c3f69bfd438ddca04d97e2f14d7b2c","first_computed_at":"2026-07-05T09:21:28.367341Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T09:21:28.367341Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"G/MeBI0Py07XyGW+uH2wKMtN7koPl1eDF/A5nL+suAchzgfB6pxK4w3zpkwfv4jiyIYxVgFJMAoArDrQjOjsBw==","signature_status":"signed_v1","signed_at":"2026-07-05T09:21:28.367839Z","signed_message":"canonical_sha256_bytes"},"source_id":"2409.15360","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:7d376341a73a2aae0e36b3297f78944a9312b60c189e15bd137393b34a9e6368","sha256:555c292cf2494faab9f2c3f207acb1e414782e4118ad2ab1e25a948fb1ae605a"],"state_sha256":"e348503a70696fec92e89a843a7ac062e6f7fd2464b7a2dff4ac8e7ad5a1324d"}