{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:L4YR4RGUOMAGXKAOI3MN2KI2OS","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"b5cb2bb7a43f2ef38b68e830eb925e9c44251a95f4798f4a9cc8dafd7b30c922","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2024-01-11T17:56:59Z","title_canon_sha256":"9bc59c8055a435abbd703a9c8d769f052216263221491a5d8d240b1fbccc3390"},"schema_version":"1.0","source":{"id":"2401.06080","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2401.06080","created_at":"2026-07-05T07:32:45Z"},{"alias_kind":"arxiv_version","alias_value":"2401.06080v2","created_at":"2026-07-05T07:32:45Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.06080","created_at":"2026-07-05T07:32:45Z"},{"alias_kind":"pith_short_12","alias_value":"L4YR4RGUOMAG","created_at":"2026-07-05T07:32:45Z"},{"alias_kind":"pith_short_16","alias_value":"L4YR4RGUOMAGXKAO","created_at":"2026-07-05T07:32:45Z"},{"alias_kind":"pith_short_8","alias_value":"L4YR4RGU","created_at":"2026-07-05T07:32:45Z"}],"graph_snapshots":[{"event_id":"sha256:2f2a45f76fd0f74448022ec86fc9109b855f2befe94e55e37b4300750bc48138","target":"graph","created_at":"2026-07-05T07:32:45Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2401.06080/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement Learning from Human Feedback (RLHF) has become a crucial technology for aligning language models with human values and intentions, enabling models to produce more helpful and harmless responses. Reward models are trained as proxies for human preferences to drive reinforcement learning optimization. While reward models are often considered central to achieving high performance, they face the following challenges in practical applications: (1) Incorrect and ambiguous preference pairs in the dataset may hinder the reward model from accurately capturing human intent. (2) Reward model","authors_text":"Binghai Wang, Caishuang Huang, Chenyu Shi, Enyu Zhou, Hang Yan, Jun Zhao, Lixing Shen, Lu Chen, Nuo Xu, Qi Zhang, Rui Zheng, Senjie Jin, Shihan Dou, Songyang Gao, Tao Gui, Tao Ji, Wei Shen, Xiaoran Fan, Xiao Wang, Xipeng Qiu, Xuanjing Huang, Yan Liu, Yu-Gang Jiang, Yuhao Zhou, Zhan Chen, Zhiheng Xi, Zuxuan Wu","cross_cats":[],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2024-01-11T17:56:59Z","title":"Secrets of RLHF in Large Language Models Part II: Reward Modeling"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.06080","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:67980136656632b9d21c4f6df3c6fdf8a27e2d0e74778626075a43ad4e30ac95","target":"record","created_at":"2026-07-05T07:32:45Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"b5cb2bb7a43f2ef38b68e830eb925e9c44251a95f4798f4a9cc8dafd7b30c922","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2024-01-11T17:56:59Z","title_canon_sha256":"9bc59c8055a435abbd703a9c8d769f052216263221491a5d8d240b1fbccc3390"},"schema_version":"1.0","source":{"id":"2401.06080","kind":"arxiv","version":2}},"canonical_sha256":"5f311e44d473006ba80e46d8dd291a7483f5f48ded40fcbb91e4e543934a54ff","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"5f311e44d473006ba80e46d8dd291a7483f5f48ded40fcbb91e4e543934a54ff","first_computed_at":"2026-07-05T07:32:45.902283Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T07:32:45.902283Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"/ThV2WbDhVSe7JfefQBIhtVSo9h/VU9oqMA8M4jo9E/CAcfof+1bczDwKuraLXtkWKvlm3BcYaXDlozKh2tCDA==","signature_status":"signed_v1","signed_at":"2026-07-05T07:32:45.902856Z","signed_message":"canonical_sha256_bytes"},"source_id":"2401.06080","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:67980136656632b9d21c4f6df3c6fdf8a27e2d0e74778626075a43ad4e30ac95","sha256:2f2a45f76fd0f74448022ec86fc9109b855f2befe94e55e37b4300750bc48138"],"state_sha256":"84bfa84223bf05b3460937000610aa29aba9fe91288c743cdf6ef5832cab7dd7"}