{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:UNOKB3IWV62HM7DNSOAP6YLO5C","short_pith_number":"pith:UNOKB3IW","canonical_record":{"source":{"id":"2601.22648","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-01-30T07:07:42Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"77eb184a66c1b46097abcf05c579cfb92144a881622ec9cd61a1d0dda6ee4ee7","abstract_canon_sha256":"f6ec9117683ef94ed146cce1581c5aa1434b9e824554ba09bebf70af16dcf120"},"schema_version":"1.0"},"canonical_sha256":"a35ca0ed16afb4767c6d9380ff616ee894a4d47a41165459bed2de39c1bab80e","source":{"kind":"arxiv","id":"2601.22648","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2601.22648","created_at":"2026-05-27T01:05:43Z"},{"alias_kind":"arxiv_version","alias_value":"2601.22648v2","created_at":"2026-05-27T01:05:43Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2601.22648","created_at":"2026-05-27T01:05:43Z"},{"alias_kind":"pith_short_12","alias_value":"UNOKB3IWV62H","created_at":"2026-05-27T01:05:43Z"},{"alias_kind":"pith_short_16","alias_value":"UNOKB3IWV62HM7DN","created_at":"2026-05-27T01:05:43Z"},{"alias_kind":"pith_short_8","alias_value":"UNOKB3IW","created_at":"2026-05-27T01:05:43Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:UNOKB3IWV62HM7DNSOAP6YLO5C","target":"record","payload":{"canonical_record":{"source":{"id":"2601.22648","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-01-30T07:07:42Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"77eb184a66c1b46097abcf05c579cfb92144a881622ec9cd61a1d0dda6ee4ee7","abstract_canon_sha256":"f6ec9117683ef94ed146cce1581c5aa1434b9e824554ba09bebf70af16dcf120"},"schema_version":"1.0"},"canonical_sha256":"a35ca0ed16afb4767c6d9380ff616ee894a4d47a41165459bed2de39c1bab80e","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-27T01:05:43.555361Z","signature_b64":"SoslIIlm2diiNsndw5a5iFYdhuEuuMhEuIT0YssGxgVcRidvbh39iwo1dXE/5yRX8e/wZMou5ID4V9Na4ETWDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a35ca0ed16afb4767c6d9380ff616ee894a4d47a41165459bed2de39c1bab80e","last_reissued_at":"2026-05-27T01:05:43.554505Z","signature_status":"signed_v1","first_computed_at":"2026-05-27T01:05:43.554505Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2601.22648","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-27T01:05:43Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"b59JA4gqZxZLttgwePFq4ATfADoxv7DaPfjJQeJqoP7I2qsfpSt4b4AdZdGASsJOAaS0MV7y/LXJqf8KbasSCA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-06-03T17:37:50.213597Z"},"content_sha256":"3f17b824c361385e207be40bd4ba6ff5b9a2f10d6d056f3379530a6e48a0d9d4","schema_version":"1.0","event_id":"sha256:3f17b824c361385e207be40bd4ba6ff5b9a2f10d6d056f3379530a6e48a0d9d4"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:UNOKB3IWV62HM7DNSOAP6YLO5C","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"UCPO: Uncertainty-Aware Policy Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Chunmei Xie, Gongrui Nan, Jing Huang, Junhao Zhang, Mengyu Lu, Qiang Zhu, Qixuan Zhou, Siye Chen, Weiqi Xiong, Xianzhou Zeng, Xingzhong Xu, Yadong Li","submitted_at":"2026-01-30T07:07:42Z","abstract_excerpt":"The key to building trustworthy large language models (LLMs) lies in endowing them with inherent uncertainty expression capabilities, thereby mitigating overconfident errors in high-stakes applications. However, existing RL paradigms such as GRPO often suffer from Advantage Bias due to binary decision spaces and static uncertainty rewards, inducing either excessive conservatism or overconfidence. To tackle this challenge, this paper unveils the root causes of reward hacking and overconfidence in current RL paradigms incorporating uncertainty-based rewards, based on which we propose the UnCerta"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2601.22648","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2601.22648/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-27T01:05:43Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"2ZPwzF2w9t8K8YgLQ/Z2M8ryOOylKhJuBrMhuZBCse1QG9+pz9zGVhTgoVm5giWNSVpr1Uw/gJVm0yB8jK1EDA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-06-03T17:37:50.213989Z"},"content_sha256":"2b49d4685d2529dc1447c5249d115f8025f699d4def2636dfdb39623a1f88715","schema_version":"1.0","event_id":"sha256:2b49d4685d2529dc1447c5249d115f8025f699d4def2636dfdb39623a1f88715"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/UNOKB3IWV62HM7DNSOAP6YLO5C/bundle.json","state_url":"https://pith.science/pith/UNOKB3IWV62HM7DNSOAP6YLO5C/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/UNOKB3IWV62HM7DNSOAP6YLO5C/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-06-03T17:37:50Z","links":{"resolver":"https://pith.science/pith/UNOKB3IWV62HM7DNSOAP6YLO5C","bundle":"https://pith.science/pith/UNOKB3IWV62HM7DNSOAP6YLO5C/bundle.json","state":"https://pith.science/pith/UNOKB3IWV62HM7DNSOAP6YLO5C/state.json","well_known_bundle":"https://pith.science/.well-known/pith/UNOKB3IWV62HM7DNSOAP6YLO5C/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:UNOKB3IWV62HM7DNSOAP6YLO5C","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"f6ec9117683ef94ed146cce1581c5aa1434b9e824554ba09bebf70af16dcf120","cross_cats_sorted":["cs.LG"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-01-30T07:07:42Z","title_canon_sha256":"77eb184a66c1b46097abcf05c579cfb92144a881622ec9cd61a1d0dda6ee4ee7"},"schema_version":"1.0","source":{"id":"2601.22648","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2601.22648","created_at":"2026-05-27T01:05:43Z"},{"alias_kind":"arxiv_version","alias_value":"2601.22648v2","created_at":"2026-05-27T01:05:43Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2601.22648","created_at":"2026-05-27T01:05:43Z"},{"alias_kind":"pith_short_12","alias_value":"UNOKB3IWV62H","created_at":"2026-05-27T01:05:43Z"},{"alias_kind":"pith_short_16","alias_value":"UNOKB3IWV62HM7DN","created_at":"2026-05-27T01:05:43Z"},{"alias_kind":"pith_short_8","alias_value":"UNOKB3IW","created_at":"2026-05-27T01:05:43Z"}],"graph_snapshots":[{"event_id":"sha256:2b49d4685d2529dc1447c5249d115f8025f699d4def2636dfdb39623a1f88715","target":"graph","created_at":"2026-05-27T01:05:43Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2601.22648/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"The key to building trustworthy large language models (LLMs) lies in endowing them with inherent uncertainty expression capabilities, thereby mitigating overconfident errors in high-stakes applications. However, existing RL paradigms such as GRPO often suffer from Advantage Bias due to binary decision spaces and static uncertainty rewards, inducing either excessive conservatism or overconfidence. To tackle this challenge, this paper unveils the root causes of reward hacking and overconfidence in current RL paradigms incorporating uncertainty-based rewards, based on which we propose the UnCerta","authors_text":"Chunmei Xie, Gongrui Nan, Jing Huang, Junhao Zhang, Mengyu Lu, Qiang Zhu, Qixuan Zhou, Siye Chen, Weiqi Xiong, Xianzhou Zeng, Xingzhong Xu, Yadong Li","cross_cats":["cs.LG"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-01-30T07:07:42Z","title":"UCPO: Uncertainty-Aware Policy Optimization"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2601.22648","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:3f17b824c361385e207be40bd4ba6ff5b9a2f10d6d056f3379530a6e48a0d9d4","target":"record","created_at":"2026-05-27T01:05:43Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"f6ec9117683ef94ed146cce1581c5aa1434b9e824554ba09bebf70af16dcf120","cross_cats_sorted":["cs.LG"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-01-30T07:07:42Z","title_canon_sha256":"77eb184a66c1b46097abcf05c579cfb92144a881622ec9cd61a1d0dda6ee4ee7"},"schema_version":"1.0","source":{"id":"2601.22648","kind":"arxiv","version":2}},"canonical_sha256":"a35ca0ed16afb4767c6d9380ff616ee894a4d47a41165459bed2de39c1bab80e","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"a35ca0ed16afb4767c6d9380ff616ee894a4d47a41165459bed2de39c1bab80e","first_computed_at":"2026-05-27T01:05:43.554505Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-05-27T01:05:43.554505Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"SoslIIlm2diiNsndw5a5iFYdhuEuuMhEuIT0YssGxgVcRidvbh39iwo1dXE/5yRX8e/wZMou5ID4V9Na4ETWDg==","signature_status":"signed_v1","signed_at":"2026-05-27T01:05:43.555361Z","signed_message":"canonical_sha256_bytes"},"source_id":"2601.22648","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:3f17b824c361385e207be40bd4ba6ff5b9a2f10d6d056f3379530a6e48a0d9d4","sha256:2b49d4685d2529dc1447c5249d115f8025f699d4def2636dfdb39623a1f88715"],"state_sha256":"2ed7e586510ab8734f60a76bae8205720b03ddf1f14083bf81ba5ea4a0bf72e9"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"6LB7SWYYZ1jz4jBbUfNmf3Q1ZdvI9jTbEiJhoWP+wtw8T15x2wmZx7MnDWVQIB+fltQT+YeDfTT3Fg/fMeIBDg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-06-03T17:37:50.216199Z","bundle_sha256":"aa1b45160c58f1307fc7486ba200cfb70fa31155d7c4265089a825318a367d21"}}