{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:JX3YDFAOKQTE7PDK2DACKLDHOZ","short_pith_number":"pith:JX3YDFAO","canonical_record":{"source":{"id":"2501.06248","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-01-08T19:03:17Z","cross_cats_sorted":["cs.AI","cs.CL","econ.GN","q-fin.EC"],"title_canon_sha256":"39904c52ea3d57ac26044b2d126c77a72380705f87a77115a602ffe5ec761668","abstract_canon_sha256":"c82c525a7be0b6478d9d71f608c78b75f3977aa2c77df28cd7c91ed9e54b7642"},"schema_version":"1.0"},"canonical_sha256":"4df781940e54264fbc6ad0c0252c677673bb4819fa3403393c73d0edadb23fcf","source":{"kind":"arxiv","id":"2501.06248","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2501.06248","created_at":"2026-07-05T10:19:35Z"},{"alias_kind":"arxiv_version","alias_value":"2501.06248v2","created_at":"2026-07-05T10:19:35Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.06248","created_at":"2026-07-05T10:19:35Z"},{"alias_kind":"pith_short_12","alias_value":"JX3YDFAOKQTE","created_at":"2026-07-05T10:19:35Z"},{"alias_kind":"pith_short_16","alias_value":"JX3YDFAOKQTE7PDK","created_at":"2026-07-05T10:19:35Z"},{"alias_kind":"pith_short_8","alias_value":"JX3YDFAO","created_at":"2026-07-05T10:19:35Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:JX3YDFAOKQTE7PDK2DACKLDHOZ","target":"record","payload":{"canonical_record":{"source":{"id":"2501.06248","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-01-08T19:03:17Z","cross_cats_sorted":["cs.AI","cs.CL","econ.GN","q-fin.EC"],"title_canon_sha256":"39904c52ea3d57ac26044b2d126c77a72380705f87a77115a602ffe5ec761668","abstract_canon_sha256":"c82c525a7be0b6478d9d71f608c78b75f3977aa2c77df28cd7c91ed9e54b7642"},"schema_version":"1.0"},"canonical_sha256":"4df781940e54264fbc6ad0c0252c677673bb4819fa3403393c73d0edadb23fcf","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:19:35.690641Z","signature_b64":"ZawAcGI2AFS7VTUCbEGHvZ/NGSUCmBnRPthihVls1S6z/kXFCC1U8ciXhKma/e6pSC4Ou/rz2TEYewMnBHH3Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4df781940e54264fbc6ad0c0252c677673bb4819fa3403393c73d0edadb23fcf","last_reissued_at":"2026-07-05T10:19:35.690206Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:19:35.690206Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2501.06248","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:19:35Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"qcvsSEFDllR7R7wTZSbJU1I9TOdZjhYv/RDfDiqURhxwpM9jMs4kY8xe9q86+GOP97MIglsTNj59lwGQCOdnBw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-13T21:31:52.851445Z"},"content_sha256":"2d0da753c70b3669b6b3794b07e5c703268858ee0888b0a19c247f071414c22c","schema_version":"1.0","event_id":"sha256:2d0da753c70b3669b6b3794b07e5c703268858ee0888b0a19c247f071414c22c"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:JX3YDFAOKQTE7PDK2DACKLDHOZ","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Utility-inspired Reward Transformations Improve Reinforcement Learning Training of Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","econ.GN","q-fin.EC"],"primary_cat":"cs.LG","authors_text":"Chirag Nagpal, Francesco Visin, Roberto-Rafael Maura-Rivero, Roma Patel","submitted_at":"2025-01-08T19:03:17Z","abstract_excerpt":"Current methods that train large language models (LLMs) with reinforcement learning feedback, often resort to averaging outputs of multiple rewards functions during training. This overlooks crucial aspects of individual reward dimensions and inter-reward dependencies that can lead to sub-optimal outcomes in generations. In this work, we show how linear aggregation of rewards exhibits some vulnerabilities that can lead to undesired properties of generated text. We then propose a transformation of reward functions inspired by economic theory of utility functions (specifically Inada conditions), "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.06248","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.06248/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:19:35Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Yeeys43PgZzzHnL5xdKU0l1QYOm+TB1BeZphH8ESUe9xDep/DxCauh6YHswffoCAFlhMQr73xnf0KjGXfAfOAw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-13T21:31:52.852109Z"},"content_sha256":"fcbfe951fa0b0336249e97bfedbbd202e2d7702c233670295f0303b378d46873","schema_version":"1.0","event_id":"sha256:fcbfe951fa0b0336249e97bfedbbd202e2d7702c233670295f0303b378d46873"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/JX3YDFAOKQTE7PDK2DACKLDHOZ/bundle.json","state_url":"https://pith.science/pith/JX3YDFAOKQTE7PDK2DACKLDHOZ/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/JX3YDFAOKQTE7PDK2DACKLDHOZ/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-13T21:31:52Z","links":{"resolver":"https://pith.science/pith/JX3YDFAOKQTE7PDK2DACKLDHOZ","bundle":"https://pith.science/pith/JX3YDFAOKQTE7PDK2DACKLDHOZ/bundle.json","state":"https://pith.science/pith/JX3YDFAOKQTE7PDK2DACKLDHOZ/state.json","well_known_bundle":"https://pith.science/.well-known/pith/JX3YDFAOKQTE7PDK2DACKLDHOZ/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:JX3YDFAOKQTE7PDK2DACKLDHOZ","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"c82c525a7be0b6478d9d71f608c78b75f3977aa2c77df28cd7c91ed9e54b7642","cross_cats_sorted":["cs.AI","cs.CL","econ.GN","q-fin.EC"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-01-08T19:03:17Z","title_canon_sha256":"39904c52ea3d57ac26044b2d126c77a72380705f87a77115a602ffe5ec761668"},"schema_version":"1.0","source":{"id":"2501.06248","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2501.06248","created_at":"2026-07-05T10:19:35Z"},{"alias_kind":"arxiv_version","alias_value":"2501.06248v2","created_at":"2026-07-05T10:19:35Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.06248","created_at":"2026-07-05T10:19:35Z"},{"alias_kind":"pith_short_12","alias_value":"JX3YDFAOKQTE","created_at":"2026-07-05T10:19:35Z"},{"alias_kind":"pith_short_16","alias_value":"JX3YDFAOKQTE7PDK","created_at":"2026-07-05T10:19:35Z"},{"alias_kind":"pith_short_8","alias_value":"JX3YDFAO","created_at":"2026-07-05T10:19:35Z"}],"graph_snapshots":[{"event_id":"sha256:fcbfe951fa0b0336249e97bfedbbd202e2d7702c233670295f0303b378d46873","target":"graph","created_at":"2026-07-05T10:19:35Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2501.06248/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Current methods that train large language models (LLMs) with reinforcement learning feedback, often resort to averaging outputs of multiple rewards functions during training. This overlooks crucial aspects of individual reward dimensions and inter-reward dependencies that can lead to sub-optimal outcomes in generations. In this work, we show how linear aggregation of rewards exhibits some vulnerabilities that can lead to undesired properties of generated text. We then propose a transformation of reward functions inspired by economic theory of utility functions (specifically Inada conditions), ","authors_text":"Chirag Nagpal, Francesco Visin, Roberto-Rafael Maura-Rivero, Roma Patel","cross_cats":["cs.AI","cs.CL","econ.GN","q-fin.EC"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-01-08T19:03:17Z","title":"Utility-inspired Reward Transformations Improve Reinforcement Learning Training of Language Models"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.06248","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:2d0da753c70b3669b6b3794b07e5c703268858ee0888b0a19c247f071414c22c","target":"record","created_at":"2026-07-05T10:19:35Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"c82c525a7be0b6478d9d71f608c78b75f3977aa2c77df28cd7c91ed9e54b7642","cross_cats_sorted":["cs.AI","cs.CL","econ.GN","q-fin.EC"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-01-08T19:03:17Z","title_canon_sha256":"39904c52ea3d57ac26044b2d126c77a72380705f87a77115a602ffe5ec761668"},"schema_version":"1.0","source":{"id":"2501.06248","kind":"arxiv","version":2}},"canonical_sha256":"4df781940e54264fbc6ad0c0252c677673bb4819fa3403393c73d0edadb23fcf","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"4df781940e54264fbc6ad0c0252c677673bb4819fa3403393c73d0edadb23fcf","first_computed_at":"2026-07-05T10:19:35.690206Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T10:19:35.690206Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"ZawAcGI2AFS7VTUCbEGHvZ/NGSUCmBnRPthihVls1S6z/kXFCC1U8ciXhKma/e6pSC4Ou/rz2TEYewMnBHH3Cg==","signature_status":"signed_v1","signed_at":"2026-07-05T10:19:35.690641Z","signed_message":"canonical_sha256_bytes"},"source_id":"2501.06248","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:2d0da753c70b3669b6b3794b07e5c703268858ee0888b0a19c247f071414c22c","sha256:fcbfe951fa0b0336249e97bfedbbd202e2d7702c233670295f0303b378d46873"],"state_sha256":"502d682c6867c14b1b7765a5eac8adfb06de64d6227bbde7ad1005b362831791"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"fbhhtxXuTGsi7x1H+X37DC98hvETJFL9YEtzF6l+4awVaHzlgQnjFDqP6lEyKbNEZoQBW3XzEcD+Q9+IxEfxDA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-13T21:31:52.869684Z","bundle_sha256":"394c756befaacc130483e7c910c882ca691a2c191a9bbb385dda0d7082ba173c"}}