{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:GFMK66JPUH2RF2S5PQ2GM5MHAS","short_pith_number":"pith:GFMK66JP","canonical_record":{"source":{"id":"2406.02764","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-06-04T20:33:22Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"a330995a3a659fc184b0c1e0110e7879b800519a0b1c82138d28f14d308dfded","abstract_canon_sha256":"666ae616d30f90878b611a95d5b4010dad22ae135fd6641f5105a9566335c036"},"schema_version":"1.0"},"canonical_sha256":"3158af792fa1f512ea5d7c3466758704aa7e10fde4212eedb5aff2e28534db79","source":{"kind":"arxiv","id":"2406.02764","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2406.02764","created_at":"2026-07-05T08:27:29Z"},{"alias_kind":"arxiv_version","alias_value":"2406.02764v1","created_at":"2026-07-05T08:27:29Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.02764","created_at":"2026-07-05T08:27:29Z"},{"alias_kind":"pith_short_12","alias_value":"GFMK66JPUH2R","created_at":"2026-07-05T08:27:29Z"},{"alias_kind":"pith_short_16","alias_value":"GFMK66JPUH2RF2S5","created_at":"2026-07-05T08:27:29Z"},{"alias_kind":"pith_short_8","alias_value":"GFMK66JP","created_at":"2026-07-05T08:27:29Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:GFMK66JPUH2RF2S5PQ2GM5MHAS","target":"record","payload":{"canonical_record":{"source":{"id":"2406.02764","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-06-04T20:33:22Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"a330995a3a659fc184b0c1e0110e7879b800519a0b1c82138d28f14d308dfded","abstract_canon_sha256":"666ae616d30f90878b611a95d5b4010dad22ae135fd6641f5105a9566335c036"},"schema_version":"1.0"},"canonical_sha256":"3158af792fa1f512ea5d7c3466758704aa7e10fde4212eedb5aff2e28534db79","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:27:29.924225Z","signature_b64":"2XImw5ghwZjQZDtbDq49Y2hqx2wFkbrpLi5JDTDcx6HPeo/UdAkaWen2Pn9a8UW9Xc52lTQsRLPBUt9MmnnnBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3158af792fa1f512ea5d7c3466758704aa7e10fde4212eedb5aff2e28534db79","last_reissued_at":"2026-07-05T08:27:29.923688Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:27:29.923688Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2406.02764","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:27:29Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"501QEwb+rCvQ/7QrP5Po3mlm/KbKx47u0Y0DbAKh36VscUr2k78nhjWh2ppgltkxiciMPxVMLiMRxMKLPPePCQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-11T02:10:47.356065Z"},"content_sha256":"68b72e8c7e7bccbf4f72f6244dafdc60bcc18caee03488c948ff1e59907be2df","schema_version":"1.0","event_id":"sha256:68b72e8c7e7bccbf4f72f6244dafdc60bcc18caee03488c948ff1e59907be2df"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:GFMK66JPUH2RF2S5PQ2GM5MHAS","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Adaptive Preference Scaling for Reinforcement Learning with Human Feedback","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Alexander Bukharin, Haoming Jiang, Ilgee Hong, Tianbao Yang, Tuo Zhao, Yixiao Li, Zichong Li","submitted_at":"2024-06-04T20:33:22Z","abstract_excerpt":"Reinforcement learning from human feedback (RLHF) is a prevalent approach to align AI systems with human values by learning rewards from human preference data. Due to various reasons, however, such data typically takes the form of rankings over pairs of trajectory segments, which fails to capture the varying strengths of preferences across different pairs. In this paper, we propose a novel adaptive preference loss, underpinned by distributionally robust optimization (DRO), designed to address this uncertainty in preference strength. By incorporating an adaptive scaling parameter into the loss "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.02764","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.02764/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:27:29Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"dYK+Rr3oiQV/3aGmVCTFWXOW4UKePQ0lKArkZHn6NLnr6QdjLwRwX98tZtD3KDIZXO/GwyO+XksJM87tbHwfAQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-11T02:10:47.356607Z"},"content_sha256":"8c347599b0253f7e2c771db5b3a0d7d912d2af98f96071dd9c6f44f10f3daad6","schema_version":"1.0","event_id":"sha256:8c347599b0253f7e2c771db5b3a0d7d912d2af98f96071dd9c6f44f10f3daad6"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/GFMK66JPUH2RF2S5PQ2GM5MHAS/bundle.json","state_url":"https://pith.science/pith/GFMK66JPUH2RF2S5PQ2GM5MHAS/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/GFMK66JPUH2RF2S5PQ2GM5MHAS/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-11T02:10:47Z","links":{"resolver":"https://pith.science/pith/GFMK66JPUH2RF2S5PQ2GM5MHAS","bundle":"https://pith.science/pith/GFMK66JPUH2RF2S5PQ2GM5MHAS/bundle.json","state":"https://pith.science/pith/GFMK66JPUH2RF2S5PQ2GM5MHAS/state.json","well_known_bundle":"https://pith.science/.well-known/pith/GFMK66JPUH2RF2S5PQ2GM5MHAS/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:GFMK66JPUH2RF2S5PQ2GM5MHAS","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"666ae616d30f90878b611a95d5b4010dad22ae135fd6641f5105a9566335c036","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-06-04T20:33:22Z","title_canon_sha256":"a330995a3a659fc184b0c1e0110e7879b800519a0b1c82138d28f14d308dfded"},"schema_version":"1.0","source":{"id":"2406.02764","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2406.02764","created_at":"2026-07-05T08:27:29Z"},{"alias_kind":"arxiv_version","alias_value":"2406.02764v1","created_at":"2026-07-05T08:27:29Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.02764","created_at":"2026-07-05T08:27:29Z"},{"alias_kind":"pith_short_12","alias_value":"GFMK66JPUH2R","created_at":"2026-07-05T08:27:29Z"},{"alias_kind":"pith_short_16","alias_value":"GFMK66JPUH2RF2S5","created_at":"2026-07-05T08:27:29Z"},{"alias_kind":"pith_short_8","alias_value":"GFMK66JP","created_at":"2026-07-05T08:27:29Z"}],"graph_snapshots":[{"event_id":"sha256:8c347599b0253f7e2c771db5b3a0d7d912d2af98f96071dd9c6f44f10f3daad6","target":"graph","created_at":"2026-07-05T08:27:29Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2406.02764/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement learning from human feedback (RLHF) is a prevalent approach to align AI systems with human values by learning rewards from human preference data. Due to various reasons, however, such data typically takes the form of rankings over pairs of trajectory segments, which fails to capture the varying strengths of preferences across different pairs. In this paper, we propose a novel adaptive preference loss, underpinned by distributionally robust optimization (DRO), designed to address this uncertainty in preference strength. By incorporating an adaptive scaling parameter into the loss ","authors_text":"Alexander Bukharin, Haoming Jiang, Ilgee Hong, Tianbao Yang, Tuo Zhao, Yixiao Li, Zichong Li","cross_cats":["cs.AI"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-06-04T20:33:22Z","title":"Adaptive Preference Scaling for Reinforcement Learning with Human Feedback"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.02764","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:68b72e8c7e7bccbf4f72f6244dafdc60bcc18caee03488c948ff1e59907be2df","target":"record","created_at":"2026-07-05T08:27:29Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"666ae616d30f90878b611a95d5b4010dad22ae135fd6641f5105a9566335c036","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-06-04T20:33:22Z","title_canon_sha256":"a330995a3a659fc184b0c1e0110e7879b800519a0b1c82138d28f14d308dfded"},"schema_version":"1.0","source":{"id":"2406.02764","kind":"arxiv","version":1}},"canonical_sha256":"3158af792fa1f512ea5d7c3466758704aa7e10fde4212eedb5aff2e28534db79","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"3158af792fa1f512ea5d7c3466758704aa7e10fde4212eedb5aff2e28534db79","first_computed_at":"2026-07-05T08:27:29.923688Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T08:27:29.923688Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"2XImw5ghwZjQZDtbDq49Y2hqx2wFkbrpLi5JDTDcx6HPeo/UdAkaWen2Pn9a8UW9Xc52lTQsRLPBUt9MmnnnBA==","signature_status":"signed_v1","signed_at":"2026-07-05T08:27:29.924225Z","signed_message":"canonical_sha256_bytes"},"source_id":"2406.02764","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:68b72e8c7e7bccbf4f72f6244dafdc60bcc18caee03488c948ff1e59907be2df","sha256:8c347599b0253f7e2c771db5b3a0d7d912d2af98f96071dd9c6f44f10f3daad6"],"state_sha256":"785d551635dc63210c122e05670153c2d39a6c651c5924d0e7f8e6f5003e3a44"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"bg3/I3+JS59b2MvWX7NMhsdzPqvHFUCqF40QSiLNDc8ZpTGVBDM8508EDFpGcj+q9UNiScZwuPzxyci+HpIZDQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-11T02:10:47.362189Z","bundle_sha256":"0cf8b9271ea2521409f71b1753d1c631c833922863bdd3b6b880e4fb4c3dc2f9"}}