{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:7AHSBTBG3FN3WOK6L7DWK32RSY","short_pith_number":"pith:7AHSBTBG","canonical_record":{"source":{"id":"2503.22480","kind":"arxiv","version":6},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-03-28T14:39:52Z","cross_cats_sorted":[],"title_canon_sha256":"d8e6b6db9c4038efe23f4b1adb9d6688b10d1d3c1de1003438d085616fe02cb0","abstract_canon_sha256":"ce3e67f24307cd6d29c07c0cedb976d74f703c1884463519546e5fbec9e07d98"},"schema_version":"1.0"},"canonical_sha256":"f80f20cc26d95bbb395e5fc7656f51962c4851921b31ea7bbb5839428963819d","source":{"kind":"arxiv","id":"2503.22480","version":6},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2503.22480","created_at":"2026-07-05T11:03:58Z"},{"alias_kind":"arxiv_version","alias_value":"2503.22480v6","created_at":"2026-07-05T11:03:58Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.22480","created_at":"2026-07-05T11:03:58Z"},{"alias_kind":"pith_short_12","alias_value":"7AHSBTBG3FN3","created_at":"2026-07-05T11:03:58Z"},{"alias_kind":"pith_short_16","alias_value":"7AHSBTBG3FN3WOK6","created_at":"2026-07-05T11:03:58Z"},{"alias_kind":"pith_short_8","alias_value":"7AHSBTBG","created_at":"2026-07-05T11:03:58Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:7AHSBTBG3FN3WOK6L7DWK32RSY","target":"record","payload":{"canonical_record":{"source":{"id":"2503.22480","kind":"arxiv","version":6},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-03-28T14:39:52Z","cross_cats_sorted":[],"title_canon_sha256":"d8e6b6db9c4038efe23f4b1adb9d6688b10d1d3c1de1003438d085616fe02cb0","abstract_canon_sha256":"ce3e67f24307cd6d29c07c0cedb976d74f703c1884463519546e5fbec9e07d98"},"schema_version":"1.0"},"canonical_sha256":"f80f20cc26d95bbb395e5fc7656f51962c4851921b31ea7bbb5839428963819d","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:03:58.763057Z","signature_b64":"UtK46sn5HbLvUnsgU3qf6hMzQdU628Kpng5DWcDLrdbl/QFyTOBcMfsYcJUsixqR+NUrTHb6I9uoYhOt+QCbDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f80f20cc26d95bbb395e5fc7656f51962c4851921b31ea7bbb5839428963819d","last_reissued_at":"2026-07-05T11:03:58.762582Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:03:58.762582Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2503.22480","source_version":6,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:03:58Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"sJp/y0kEhv+wcBiNOWXPyz/wPCvj+NFzi7kcb1CBm4IY0xlIt8HXVc2lWbiwP+SMXYdUeNaj2xvz7aGUFB+PDA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T04:17:49.241037Z"},"content_sha256":"fba6c9e5bc7dc361e300afd8c837730d6e9d189b0be1991c44dd428633454b1d","schema_version":"1.0","event_id":"sha256:fba6c9e5bc7dc361e300afd8c837730d6e9d189b0be1991c44dd428633454b1d"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:7AHSBTBG3FN3WOK6L7DWK32RSY","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Probabilistic Uncertain Reward Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Haotian Xu, Jun Zhao, Kang Liu, Shizhu He, Wangtao Sun, Xiang Cheng, Xing Yu, Zhao Yang","submitted_at":"2025-03-28T14:39:52Z","abstract_excerpt":"Reinforcement learning from human feedback (RLHF) is a critical technique for training large language models. However, conventional reward models based on the Bradley-Terry model (BTRM) often suffer from overconfidence when faced with inconsistent labels or out-of-distribution samples, leading to reward hacking, where the policy model blindly optimizes for proxy rewards while degrading true performance.\n  This paper proposes the Probabilistic Uncertain Reward Model (PURM), which generalizes the Bradley-Terry model to learn the reward distributions that emerged from the preference data. We theo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.22480","kind":"arxiv","version":6},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.22480/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:03:58Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"JsIl4u0PaAFx/OR4FQN9miGW/0dy5E3FPmwRVCbTpcKbwZoKqlWezYb6hpk88YnbThOjmxTetSX7HBoM+S0GAw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T04:17:49.241795Z"},"content_sha256":"bfd612deb81280ccf48eab2bdc66a4e7543277112c9bfcf6b50293d3d5b71679","schema_version":"1.0","event_id":"sha256:bfd612deb81280ccf48eab2bdc66a4e7543277112c9bfcf6b50293d3d5b71679"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/7AHSBTBG3FN3WOK6L7DWK32RSY/bundle.json","state_url":"https://pith.science/pith/7AHSBTBG3FN3WOK6L7DWK32RSY/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/7AHSBTBG3FN3WOK6L7DWK32RSY/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-09T04:17:49Z","links":{"resolver":"https://pith.science/pith/7AHSBTBG3FN3WOK6L7DWK32RSY","bundle":"https://pith.science/pith/7AHSBTBG3FN3WOK6L7DWK32RSY/bundle.json","state":"https://pith.science/pith/7AHSBTBG3FN3WOK6L7DWK32RSY/state.json","well_known_bundle":"https://pith.science/.well-known/pith/7AHSBTBG3FN3WOK6L7DWK32RSY/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:7AHSBTBG3FN3WOK6L7DWK32RSY","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"ce3e67f24307cd6d29c07c0cedb976d74f703c1884463519546e5fbec9e07d98","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-03-28T14:39:52Z","title_canon_sha256":"d8e6b6db9c4038efe23f4b1adb9d6688b10d1d3c1de1003438d085616fe02cb0"},"schema_version":"1.0","source":{"id":"2503.22480","kind":"arxiv","version":6}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2503.22480","created_at":"2026-07-05T11:03:58Z"},{"alias_kind":"arxiv_version","alias_value":"2503.22480v6","created_at":"2026-07-05T11:03:58Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.22480","created_at":"2026-07-05T11:03:58Z"},{"alias_kind":"pith_short_12","alias_value":"7AHSBTBG3FN3","created_at":"2026-07-05T11:03:58Z"},{"alias_kind":"pith_short_16","alias_value":"7AHSBTBG3FN3WOK6","created_at":"2026-07-05T11:03:58Z"},{"alias_kind":"pith_short_8","alias_value":"7AHSBTBG","created_at":"2026-07-05T11:03:58Z"}],"graph_snapshots":[{"event_id":"sha256:bfd612deb81280ccf48eab2bdc66a4e7543277112c9bfcf6b50293d3d5b71679","target":"graph","created_at":"2026-07-05T11:03:58Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2503.22480/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement learning from human feedback (RLHF) is a critical technique for training large language models. However, conventional reward models based on the Bradley-Terry model (BTRM) often suffer from overconfidence when faced with inconsistent labels or out-of-distribution samples, leading to reward hacking, where the policy model blindly optimizes for proxy rewards while degrading true performance.\n  This paper proposes the Probabilistic Uncertain Reward Model (PURM), which generalizes the Bradley-Terry model to learn the reward distributions that emerged from the preference data. We theo","authors_text":"Haotian Xu, Jun Zhao, Kang Liu, Shizhu He, Wangtao Sun, Xiang Cheng, Xing Yu, Zhao Yang","cross_cats":[],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-03-28T14:39:52Z","title":"Probabilistic Uncertain Reward Model"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.22480","kind":"arxiv","version":6},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:fba6c9e5bc7dc361e300afd8c837730d6e9d189b0be1991c44dd428633454b1d","target":"record","created_at":"2026-07-05T11:03:58Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"ce3e67f24307cd6d29c07c0cedb976d74f703c1884463519546e5fbec9e07d98","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-03-28T14:39:52Z","title_canon_sha256":"d8e6b6db9c4038efe23f4b1adb9d6688b10d1d3c1de1003438d085616fe02cb0"},"schema_version":"1.0","source":{"id":"2503.22480","kind":"arxiv","version":6}},"canonical_sha256":"f80f20cc26d95bbb395e5fc7656f51962c4851921b31ea7bbb5839428963819d","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"f80f20cc26d95bbb395e5fc7656f51962c4851921b31ea7bbb5839428963819d","first_computed_at":"2026-07-05T11:03:58.762582Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:03:58.762582Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"UtK46sn5HbLvUnsgU3qf6hMzQdU628Kpng5DWcDLrdbl/QFyTOBcMfsYcJUsixqR+NUrTHb6I9uoYhOt+QCbDw==","signature_status":"signed_v1","signed_at":"2026-07-05T11:03:58.763057Z","signed_message":"canonical_sha256_bytes"},"source_id":"2503.22480","source_kind":"arxiv","source_version":6}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:fba6c9e5bc7dc361e300afd8c837730d6e9d189b0be1991c44dd428633454b1d","sha256:bfd612deb81280ccf48eab2bdc66a4e7543277112c9bfcf6b50293d3d5b71679"],"state_sha256":"fb55aa7fadc8f3cfcda6d19ac206e3b0f208cc72e251cfa4074b73e770e33e9d"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"emJxtPygYSzhCWwrqwG3mMCeWGjwBajdtECPuJbnIhyVVXKShqJs6uKKxkVKo6MziY7lPtWBpzIRenJbPMM/DQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-09T04:17:49.248109Z","bundle_sha256":"f5c97badf78495d1a362d0fe953172b40fd461861cdabb49cfb38c579062772e"}}