{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:QYZC66WYDIWOKVDUNSVXWMC3G7","short_pith_number":"pith:QYZC66WY","schema_version":"1.0","canonical_sha256":"86322f7ad81a2ce554746cab7b305b37ce2986861d804b2f2086f0abaf505363","source":{"kind":"arxiv","id":"2406.12845","version":1},"attestation_state":"computed","paper":{"title":"Interpretable Preferences via Multi-Objective Reward Modeling and Mixture-of-Experts","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Han Zhao, Haoxiang Wang, Tengyang Xie, Tong Zhang, Wei Xiong","submitted_at":"2024-06-18T17:58:28Z","abstract_excerpt":"Reinforcement learning from human feedback (RLHF) has emerged as the primary method for aligning large language models (LLMs) with human preferences. The RLHF process typically starts by training a reward model (RM) using human preference data. Conventional RMs are trained on pairwise responses to the same user request, with relative ratings indicating which response humans prefer. The trained RM serves as a proxy for human preferences. However, due to the black-box nature of RMs, their outputs lack interpretability, as humans cannot intuitively understand why an RM thinks a response is good o"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.12845","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-06-18T17:58:28Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"b9f56a6e930ab8286c35aae39ecfcd1ea8b8c9a6b9283dd658b783c258113562","abstract_canon_sha256":"a5025b0bb5802f9d17e4226f19c7a04cbb00722886cfaefe954823a0c8e2bdd2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:33:51.912615Z","signature_b64":"3XuMIfNz6gMNpQ7XfurSJvxwSDv8l5j97NxogW9r3TyvxG7FzQQHS8sr5m6R/eM1yCu59O1OBIF2KlBMciqtAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"86322f7ad81a2ce554746cab7b305b37ce2986861d804b2f2086f0abaf505363","last_reissued_at":"2026-07-05T08:33:51.912118Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:33:51.912118Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Interpretable Preferences via Multi-Objective Reward Modeling and Mixture-of-Experts","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Han Zhao, Haoxiang Wang, Tengyang Xie, Tong Zhang, Wei Xiong","submitted_at":"2024-06-18T17:58:28Z","abstract_excerpt":"Reinforcement learning from human feedback (RLHF) has emerged as the primary method for aligning large language models (LLMs) with human preferences. The RLHF process typically starts by training a reward model (RM) using human preference data. Conventional RMs are trained on pairwise responses to the same user request, with relative ratings indicating which response humans prefer. The trained RM serves as a proxy for human preferences. However, due to the black-box nature of RMs, their outputs lack interpretability, as humans cannot intuitively understand why an RM thinks a response is good o"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.12845","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.12845/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.12845","created_at":"2026-07-05T08:33:51.912180+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.12845v1","created_at":"2026-07-05T08:33:51.912180+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.12845","created_at":"2026-07-05T08:33:51.912180+00:00"},{"alias_kind":"pith_short_12","alias_value":"QYZC66WYDIWO","created_at":"2026-07-05T08:33:51.912180+00:00"},{"alias_kind":"pith_short_16","alias_value":"QYZC66WYDIWOKVDU","created_at":"2026-07-05T08:33:51.912180+00:00"},{"alias_kind":"pith_short_8","alias_value":"QYZC66WY","created_at":"2026-07-05T08:33:51.912180+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26917","citing_title":"GEOALIGN: Geometric Rollout Curation for Robust LLM Reinforcement Learning","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2506.01937","citing_title":"RewardBench 2: Advancing Reward Model Evaluation","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2601.21619","citing_title":"On the Overscaling Curse of Parallel Thinking: System Efficacy Contradicts Sample Efficiency","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11679","citing_title":"Explaining and Breaking the Safety-Helpfulness Ceiling via Preference Dimensional Expansion","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02906","citing_title":"OpsLLM: Construction of Large Language Model for Software Operations with Multi-stage Learning","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11679","citing_title":"Explaining and Breaking the Safety-Helpfulness Ceiling via Preference Dimensional Expansion","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25872","citing_title":"When Errors Can Be Beneficial: A Categorization of Imperfect Rewards for Policy Gradient","ref_index":88,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02906","citing_title":"OpsLLM: Construction of Large Language Model for Software Operations with Multi-stage Learning","ref_index":45,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QYZC66WYDIWOKVDUNSVXWMC3G7","json":"https://pith.science/pith/QYZC66WYDIWOKVDUNSVXWMC3G7.json","graph_json":"https://pith.science/api/pith-number/QYZC66WYDIWOKVDUNSVXWMC3G7/graph.json","events_json":"https://pith.science/api/pith-number/QYZC66WYDIWOKVDUNSVXWMC3G7/events.json","paper":"https://pith.science/paper/QYZC66WY"},"agent_actions":{"view_html":"https://pith.science/pith/QYZC66WYDIWOKVDUNSVXWMC3G7","download_json":"https://pith.science/pith/QYZC66WYDIWOKVDUNSVXWMC3G7.json","view_paper":"https://pith.science/paper/QYZC66WY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.12845&json=true","fetch_graph":"https://pith.science/api/pith-number/QYZC66WYDIWOKVDUNSVXWMC3G7/graph.json","fetch_events":"https://pith.science/api/pith-number/QYZC66WYDIWOKVDUNSVXWMC3G7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QYZC66WYDIWOKVDUNSVXWMC3G7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QYZC66WYDIWOKVDUNSVXWMC3G7/action/storage_attestation","attest_author":"https://pith.science/pith/QYZC66WYDIWOKVDUNSVXWMC3G7/action/author_attestation","sign_citation":"https://pith.science/pith/QYZC66WYDIWOKVDUNSVXWMC3G7/action/citation_signature","submit_replication":"https://pith.science/pith/QYZC66WYDIWOKVDUNSVXWMC3G7/action/replication_record"}},"created_at":"2026-07-05T08:33:51.912180+00:00","updated_at":"2026-07-05T08:33:51.912180+00:00"}