{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:YKI4JOUUOKNSKZHJKFOSJHOOZZ","short_pith_number":"pith:YKI4JOUU","schema_version":"1.0","canonical_sha256":"c291c4ba94729b2564e9515d249dcece46b585cf2a5ad5998db9de13a3e440e0","source":{"kind":"arxiv","id":"2402.00742","version":2},"attestation_state":"computed","paper":{"title":"Transforming and Combining Rewards for Aligning Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Alex D'Amour, Chirag Nagpal, Jacob Eisenstein, Jonathan Berant, Sanmi Koyejo, Victor Veitch, Zihao Wang","submitted_at":"2024-02-01T16:39:28Z","abstract_excerpt":"A common approach for aligning language models to human preferences is to first learn a reward model from preference data, and then use this reward model to update the language model. We study two closely related problems that arise in this approach. First, any monotone transformation of the reward model preserves preference ranking; is there a choice that is ``better'' than others? Second, we often wish to align language models to multiple properties: how should we combine multiple reward models? Using a probabilistic interpretation of the alignment procedure, we identify a natural choice for"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.00742","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-02-01T16:39:28Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"25da635e53ed4ecf84da86c4cc307e90124b13a554eb0fa05d0cfb8dda896ea6","abstract_canon_sha256":"6aff8462b2d6ede8c1b86cda4b9bb747470734951cbef06ad8ac19d1ab2879d3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:45:49.867809Z","signature_b64":"BSpWfHRHiiWEKObP4qYLi0XdFVEb35NohzniUoPxmbue/IS8UfvClV8wu60WKHMszMTKgpFMLnIu5tfTr0BzDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c291c4ba94729b2564e9515d249dcece46b585cf2a5ad5998db9de13a3e440e0","last_reissued_at":"2026-07-05T08:45:49.867335Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:45:49.867335Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Transforming and Combining Rewards for Aligning Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Alex D'Amour, Chirag Nagpal, Jacob Eisenstein, Jonathan Berant, Sanmi Koyejo, Victor Veitch, Zihao Wang","submitted_at":"2024-02-01T16:39:28Z","abstract_excerpt":"A common approach for aligning language models to human preferences is to first learn a reward model from preference data, and then use this reward model to update the language model. We study two closely related problems that arise in this approach. First, any monotone transformation of the reward model preserves preference ranking; is there a choice that is ``better'' than others? Second, we often wish to align language models to multiple properties: how should we combine multiple reward models? Using a probabilistic interpretation of the alignment procedure, we identify a natural choice for"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.00742","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.00742/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.00742","created_at":"2026-07-05T08:45:49.867392+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.00742v2","created_at":"2026-07-05T08:45:49.867392+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.00742","created_at":"2026-07-05T08:45:49.867392+00:00"},{"alias_kind":"pith_short_12","alias_value":"YKI4JOUUOKNS","created_at":"2026-07-05T08:45:49.867392+00:00"},{"alias_kind":"pith_short_16","alias_value":"YKI4JOUUOKNSKZHJ","created_at":"2026-07-05T08:45:49.867392+00:00"},{"alias_kind":"pith_short_8","alias_value":"YKI4JOUU","created_at":"2026-07-05T08:45:49.867392+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12360","citing_title":"Anatomy of Post-Training: Using Interpretability to Characterize Data and Shape the Learning Signal","ref_index":99,"is_internal_anchor":false},{"citing_arxiv_id":"2505.12741","citing_title":"Language Model Networks: Supervision-Efficient Learning through Dense Communication","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2406.02430","citing_title":"Seed-TTS: A Family of High-Quality Versatile Speech Generation Models","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12474","citing_title":"Reward Hacking in Rubric-Based Reinforcement Learning","ref_index":33,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YKI4JOUUOKNSKZHJKFOSJHOOZZ","json":"https://pith.science/pith/YKI4JOUUOKNSKZHJKFOSJHOOZZ.json","graph_json":"https://pith.science/api/pith-number/YKI4JOUUOKNSKZHJKFOSJHOOZZ/graph.json","events_json":"https://pith.science/api/pith-number/YKI4JOUUOKNSKZHJKFOSJHOOZZ/events.json","paper":"https://pith.science/paper/YKI4JOUU"},"agent_actions":{"view_html":"https://pith.science/pith/YKI4JOUUOKNSKZHJKFOSJHOOZZ","download_json":"https://pith.science/pith/YKI4JOUUOKNSKZHJKFOSJHOOZZ.json","view_paper":"https://pith.science/paper/YKI4JOUU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.00742&json=true","fetch_graph":"https://pith.science/api/pith-number/YKI4JOUUOKNSKZHJKFOSJHOOZZ/graph.json","fetch_events":"https://pith.science/api/pith-number/YKI4JOUUOKNSKZHJKFOSJHOOZZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YKI4JOUUOKNSKZHJKFOSJHOOZZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YKI4JOUUOKNSKZHJKFOSJHOOZZ/action/storage_attestation","attest_author":"https://pith.science/pith/YKI4JOUUOKNSKZHJKFOSJHOOZZ/action/author_attestation","sign_citation":"https://pith.science/pith/YKI4JOUUOKNSKZHJKFOSJHOOZZ/action/citation_signature","submit_replication":"https://pith.science/pith/YKI4JOUUOKNSKZHJKFOSJHOOZZ/action/replication_record"}},"created_at":"2026-07-05T08:45:49.867392+00:00","updated_at":"2026-07-05T08:45:49.867392+00:00"}