{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:O5Q2STG23PLJVSM7CWFT7Q5UXD","short_pith_number":"pith:O5Q2STG2","schema_version":"1.0","canonical_sha256":"7761a94cdadbd69ac99f158b3fc3b4b8cce7621819ba3bfe38f5bcad94783c78","source":{"kind":"arxiv","id":"2505.10597","version":2},"attestation_state":"computed","paper":{"title":"Two Minds Better Than One: Collaborative Reward Modeling for LLM Alignment","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Jiahuan Li, Jiazheng Zhang, Jingang Wang, Mingxu Chai, Qi Zhang, Rongxiang Weng, Shibo Hong, Shihan Dou, Tao Gui, Wenqing Jing, Zhiheng Xi, Zizhuo Zhang","submitted_at":"2025-05-15T10:58:20Z","abstract_excerpt":"Reward models (RMs) play a pivotal role in aligning large language models (LLMs) with human values. However, noisy preferences in human feedback can lead to reward misgeneralization - a phenomenon where reward models learn spurious correlations or overfit to noisy preferences, which poses important challenges to the generalization of RMs. This paper systematically analyzes the characteristics of preference pairs and aims to identify how noisy preferences differ from human-aligned preferences in reward modeling. Our analysis reveals that noisy preferences are difficult for RMs to fit, as they c"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.10597","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-05-15T10:58:20Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"dd8dcda5ea428cca9aa8d4d27a10dd1044ecc4b869eca931f6afa813d7d2b6dd","abstract_canon_sha256":"3011aa46b06ce529deaabf33efa132bcb0dcbd85eea7982032030239fbb76fbc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:04:52.162227Z","signature_b64":"0vS4vdFe8JhovskGPJdiHCVIIQLZZphQgNXAIvGX2mkNneiWzAkvvslz1d7lWONxItUk6sgoOqZAn6jK/HL8Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7761a94cdadbd69ac99f158b3fc3b4b8cce7621819ba3bfe38f5bcad94783c78","last_reissued_at":"2026-07-05T11:04:52.161709Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:04:52.161709Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Two Minds Better Than One: Collaborative Reward Modeling for LLM Alignment","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Jiahuan Li, Jiazheng Zhang, Jingang Wang, Mingxu Chai, Qi Zhang, Rongxiang Weng, Shibo Hong, Shihan Dou, Tao Gui, Wenqing Jing, Zhiheng Xi, Zizhuo Zhang","submitted_at":"2025-05-15T10:58:20Z","abstract_excerpt":"Reward models (RMs) play a pivotal role in aligning large language models (LLMs) with human values. However, noisy preferences in human feedback can lead to reward misgeneralization - a phenomenon where reward models learn spurious correlations or overfit to noisy preferences, which poses important challenges to the generalization of RMs. This paper systematically analyzes the characteristics of preference pairs and aims to identify how noisy preferences differ from human-aligned preferences in reward modeling. Our analysis reveals that noisy preferences are difficult for RMs to fit, as they c"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.10597","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.10597/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.10597","created_at":"2026-07-05T11:04:52.161768+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.10597v2","created_at":"2026-07-05T11:04:52.161768+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.10597","created_at":"2026-07-05T11:04:52.161768+00:00"},{"alias_kind":"pith_short_12","alias_value":"O5Q2STG23PLJ","created_at":"2026-07-05T11:04:52.161768+00:00"},{"alias_kind":"pith_short_16","alias_value":"O5Q2STG23PLJVSM7","created_at":"2026-07-05T11:04:52.161768+00:00"},{"alias_kind":"pith_short_8","alias_value":"O5Q2STG2","created_at":"2026-07-05T11:04:52.161768+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.16004","citing_title":"AgentV-RL: Scaling Reward Modeling with Agentic Verifier","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/O5Q2STG23PLJVSM7CWFT7Q5UXD","json":"https://pith.science/pith/O5Q2STG23PLJVSM7CWFT7Q5UXD.json","graph_json":"https://pith.science/api/pith-number/O5Q2STG23PLJVSM7CWFT7Q5UXD/graph.json","events_json":"https://pith.science/api/pith-number/O5Q2STG23PLJVSM7CWFT7Q5UXD/events.json","paper":"https://pith.science/paper/O5Q2STG2"},"agent_actions":{"view_html":"https://pith.science/pith/O5Q2STG23PLJVSM7CWFT7Q5UXD","download_json":"https://pith.science/pith/O5Q2STG23PLJVSM7CWFT7Q5UXD.json","view_paper":"https://pith.science/paper/O5Q2STG2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.10597&json=true","fetch_graph":"https://pith.science/api/pith-number/O5Q2STG23PLJVSM7CWFT7Q5UXD/graph.json","fetch_events":"https://pith.science/api/pith-number/O5Q2STG23PLJVSM7CWFT7Q5UXD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/O5Q2STG23PLJVSM7CWFT7Q5UXD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/O5Q2STG23PLJVSM7CWFT7Q5UXD/action/storage_attestation","attest_author":"https://pith.science/pith/O5Q2STG23PLJVSM7CWFT7Q5UXD/action/author_attestation","sign_citation":"https://pith.science/pith/O5Q2STG23PLJVSM7CWFT7Q5UXD/action/citation_signature","submit_replication":"https://pith.science/pith/O5Q2STG23PLJVSM7CWFT7Q5UXD/action/replication_record"}},"created_at":"2026-07-05T11:04:52.161768+00:00","updated_at":"2026-07-05T11:04:52.161768+00:00"}