{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:RUDLKCMMNR7L6KJOPF4DLYUWHW","short_pith_number":"pith:RUDLKCMM","schema_version":"1.0","canonical_sha256":"8d06b5098c6c7ebf292e797835e2963da4f1fe2cd8472d6732fd7fabcec00fb2","source":{"kind":"arxiv","id":"2309.03126","version":2},"attestation_state":"computed","paper":{"title":"Everyone Deserves A Reward: Learning Customized Human Preferences","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jiawen Xie, Ke Bai, Nan Du, Pengyu Cheng, Yong Dai","submitted_at":"2023-09-06T16:03:59Z","abstract_excerpt":"Reward models (RMs) are essential for aligning large language models (LLMs) with human preferences to improve interaction quality. However, the real world is pluralistic, which leads to diversified human preferences with respect to different religions, politics, cultures, etc. Moreover, each individual can have their unique preferences on various topics. Neglecting the diversity of human preferences, current human feedback aligning methods only consider a general reward model, which is below satisfaction for customized or personalized application scenarios. To explore customized preference lea"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.03126","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-09-06T16:03:59Z","cross_cats_sorted":[],"title_canon_sha256":"ac4d66b9a4339a1b5501db92fb3345cec88dea1bda947deb716dcd01fa403e51","abstract_canon_sha256":"70f4a48c7aa46a668b83c9af792d40155cb6fb82176b49c018693d607b14a512"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:50:58.314385Z","signature_b64":"DcZFik7tEEUDyYMDZh9s8WNS2rsBcYC+Nkx/0mPzNrNMZCeocCk9AQx3vxuRLhPXeOxrzZvW/SHRwJkLYqPuDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8d06b5098c6c7ebf292e797835e2963da4f1fe2cd8472d6732fd7fabcec00fb2","last_reissued_at":"2026-07-05T06:50:58.313760Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:50:58.313760Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Everyone Deserves A Reward: Learning Customized Human Preferences","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jiawen Xie, Ke Bai, Nan Du, Pengyu Cheng, Yong Dai","submitted_at":"2023-09-06T16:03:59Z","abstract_excerpt":"Reward models (RMs) are essential for aligning large language models (LLMs) with human preferences to improve interaction quality. However, the real world is pluralistic, which leads to diversified human preferences with respect to different religions, politics, cultures, etc. Moreover, each individual can have their unique preferences on various topics. Neglecting the diversity of human preferences, current human feedback aligning methods only consider a general reward model, which is below satisfaction for customized or personalized application scenarios. To explore customized preference lea"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.03126","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.03126/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.03126","created_at":"2026-07-05T06:50:58.313833+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.03126v2","created_at":"2026-07-05T06:50:58.313833+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.03126","created_at":"2026-07-05T06:50:58.313833+00:00"},{"alias_kind":"pith_short_12","alias_value":"RUDLKCMMNR7L","created_at":"2026-07-05T06:50:58.313833+00:00"},{"alias_kind":"pith_short_16","alias_value":"RUDLKCMMNR7L6KJO","created_at":"2026-07-05T06:50:58.313833+00:00"},{"alias_kind":"pith_short_8","alias_value":"RUDLKCMM","created_at":"2026-07-05T06:50:58.313833+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07988","citing_title":"PAFO: Pareto Fairness Optimization for Personalized Reward Modeling","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03980","citing_title":"Skill-RM: Unifying Heterogeneous Evaluation Criteria via Agent Skill","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RUDLKCMMNR7L6KJOPF4DLYUWHW","json":"https://pith.science/pith/RUDLKCMMNR7L6KJOPF4DLYUWHW.json","graph_json":"https://pith.science/api/pith-number/RUDLKCMMNR7L6KJOPF4DLYUWHW/graph.json","events_json":"https://pith.science/api/pith-number/RUDLKCMMNR7L6KJOPF4DLYUWHW/events.json","paper":"https://pith.science/paper/RUDLKCMM"},"agent_actions":{"view_html":"https://pith.science/pith/RUDLKCMMNR7L6KJOPF4DLYUWHW","download_json":"https://pith.science/pith/RUDLKCMMNR7L6KJOPF4DLYUWHW.json","view_paper":"https://pith.science/paper/RUDLKCMM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.03126&json=true","fetch_graph":"https://pith.science/api/pith-number/RUDLKCMMNR7L6KJOPF4DLYUWHW/graph.json","fetch_events":"https://pith.science/api/pith-number/RUDLKCMMNR7L6KJOPF4DLYUWHW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RUDLKCMMNR7L6KJOPF4DLYUWHW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RUDLKCMMNR7L6KJOPF4DLYUWHW/action/storage_attestation","attest_author":"https://pith.science/pith/RUDLKCMMNR7L6KJOPF4DLYUWHW/action/author_attestation","sign_citation":"https://pith.science/pith/RUDLKCMMNR7L6KJOPF4DLYUWHW/action/citation_signature","submit_replication":"https://pith.science/pith/RUDLKCMMNR7L6KJOPF4DLYUWHW/action/replication_record"}},"created_at":"2026-07-05T06:50:58.313833+00:00","updated_at":"2026-07-05T06:50:58.313833+00:00"}