{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:HCGP6Q36KR4DJZEWUB3RPTDFLR","short_pith_number":"pith:HCGP6Q36","schema_version":"1.0","canonical_sha256":"388cff437e547834e496a07717cc655c5ee6466b0385ebfdc3d5f3d5e841c9ff","source":{"kind":"arxiv","id":"2410.09893","version":2},"attestation_state":"computed","paper":{"title":"RMB: Comprehensively Benchmarking Reward Models in LLM Alignment","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Binghai Wang, Enyu Zhou, Guodong Zheng, Jessica Fan, Limao Xiong, Qi Zhang, Rong Bao, Rui Zheng, Shihan Dou, Tao Gui, Wei Shen, Xuanjing Huang, Yurong Mou, Zhiheng Xi","submitted_at":"2024-10-13T16:06:54Z","abstract_excerpt":"Reward models (RMs) guide the alignment of large language models (LLMs), steering them toward behaviors preferred by humans. Evaluating RMs is the key to better aligning LLMs. However, the current evaluation of RMs may not directly correspond to their alignment performance due to the limited distribution of evaluation data and evaluation methods that are not closely related to alignment objectives. To address these limitations, we propose RMB, a comprehensive RM benchmark that covers over 49 real-world scenarios and includes both pairwise and Best-of-N (BoN) evaluations to better reflect the e"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.09893","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-13T16:06:54Z","cross_cats_sorted":[],"title_canon_sha256":"f207df1c36362b0416b17ac01f6e3494489996ffa94e22721e507c199253859a","abstract_canon_sha256":"93bb17d9c1324c78e7c1ce5398c8c6826b1f4e47d4e8a17a6f472d4840aa7eb0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:44:18.445164Z","signature_b64":"vubfw5OSUsYZKzuqlV/bpSs9xkbTb48+do/nGfSE9CLDzCb7cyetCwaHzUZ6mMVcm56IhiuK9jAPLnRT7HvhBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"388cff437e547834e496a07717cc655c5ee6466b0385ebfdc3d5f3d5e841c9ff","last_reissued_at":"2026-07-05T10:44:18.444671Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:44:18.444671Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RMB: Comprehensively Benchmarking Reward Models in LLM Alignment","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Binghai Wang, Enyu Zhou, Guodong Zheng, Jessica Fan, Limao Xiong, Qi Zhang, Rong Bao, Rui Zheng, Shihan Dou, Tao Gui, Wei Shen, Xuanjing Huang, Yurong Mou, Zhiheng Xi","submitted_at":"2024-10-13T16:06:54Z","abstract_excerpt":"Reward models (RMs) guide the alignment of large language models (LLMs), steering them toward behaviors preferred by humans. Evaluating RMs is the key to better aligning LLMs. However, the current evaluation of RMs may not directly correspond to their alignment performance due to the limited distribution of evaluation data and evaluation methods that are not closely related to alignment objectives. To address these limitations, we propose RMB, a comprehensive RM benchmark that covers over 49 real-world scenarios and includes both pairwise and Best-of-N (BoN) evaluations to better reflect the e"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.09893","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.09893/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.09893","created_at":"2026-07-05T10:44:18.444729+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.09893v2","created_at":"2026-07-05T10:44:18.444729+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.09893","created_at":"2026-07-05T10:44:18.444729+00:00"},{"alias_kind":"pith_short_12","alias_value":"HCGP6Q36KR4D","created_at":"2026-07-05T10:44:18.444729+00:00"},{"alias_kind":"pith_short_16","alias_value":"HCGP6Q36KR4DJZEW","created_at":"2026-07-05T10:44:18.444729+00:00"},{"alias_kind":"pith_short_8","alias_value":"HCGP6Q36","created_at":"2026-07-05T10:44:18.444729+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.22643","citing_title":"Boiling the Frog: A Multi-Turn Benchmark for Agentic Safety","ref_index":100,"is_internal_anchor":false},{"citing_arxiv_id":"2412.15115","citing_title":"Qwen2.5 Technical Report","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22643","citing_title":"Boiling the Frog: A Multi-Turn Benchmark for Agentic Safety","ref_index":100,"is_internal_anchor":false},{"citing_arxiv_id":"2506.01937","citing_title":"RewardBench 2: Advancing Reward Model Evaluation","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01831","citing_title":"RMGAP: Benchmarking the Generalization of Reward Models across Diverse Preferences","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07343","citing_title":"Personalized RewardBench: Evaluating Reward Models with Human Aligned Personalization","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HCGP6Q36KR4DJZEWUB3RPTDFLR","json":"https://pith.science/pith/HCGP6Q36KR4DJZEWUB3RPTDFLR.json","graph_json":"https://pith.science/api/pith-number/HCGP6Q36KR4DJZEWUB3RPTDFLR/graph.json","events_json":"https://pith.science/api/pith-number/HCGP6Q36KR4DJZEWUB3RPTDFLR/events.json","paper":"https://pith.science/paper/HCGP6Q36"},"agent_actions":{"view_html":"https://pith.science/pith/HCGP6Q36KR4DJZEWUB3RPTDFLR","download_json":"https://pith.science/pith/HCGP6Q36KR4DJZEWUB3RPTDFLR.json","view_paper":"https://pith.science/paper/HCGP6Q36","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.09893&json=true","fetch_graph":"https://pith.science/api/pith-number/HCGP6Q36KR4DJZEWUB3RPTDFLR/graph.json","fetch_events":"https://pith.science/api/pith-number/HCGP6Q36KR4DJZEWUB3RPTDFLR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HCGP6Q36KR4DJZEWUB3RPTDFLR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HCGP6Q36KR4DJZEWUB3RPTDFLR/action/storage_attestation","attest_author":"https://pith.science/pith/HCGP6Q36KR4DJZEWUB3RPTDFLR/action/author_attestation","sign_citation":"https://pith.science/pith/HCGP6Q36KR4DJZEWUB3RPTDFLR/action/citation_signature","submit_replication":"https://pith.science/pith/HCGP6Q36KR4DJZEWUB3RPTDFLR/action/replication_record"}},"created_at":"2026-07-05T10:44:18.444729+00:00","updated_at":"2026-07-05T10:44:18.444729+00:00"}