{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:CQ5BEX5V4NLRB3XKK4M2OV47VR","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"69b798fffc7b00fd2418675618aa5bd9dbba0e934d2b7347c0572aefb61b7e45","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-28T17:59:03Z","title_canon_sha256":"a38b8656c97931ad1c88b8abf5a587952a0d2f546fa50a71a6e9e5f5f35537de"},"schema_version":"1.0","source":{"id":"2505.22653","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2505.22653","created_at":"2026-07-05T11:11:28Z"},{"alias_kind":"arxiv_version","alias_value":"2505.22653v1","created_at":"2026-07-05T11:11:28Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.22653","created_at":"2026-07-05T11:11:28Z"},{"alias_kind":"pith_short_12","alias_value":"CQ5BEX5V4NLR","created_at":"2026-07-05T11:11:28Z"},{"alias_kind":"pith_short_16","alias_value":"CQ5BEX5V4NLRB3XK","created_at":"2026-07-05T11:11:28Z"},{"alias_kind":"pith_short_8","alias_value":"CQ5BEX5V","created_at":"2026-07-05T11:11:28Z"}],"graph_snapshots":[{"event_id":"sha256:c95c1d84afbdedd5c78940d00abd589df807d1579fabd0668c37b10a745164e1","target":"graph","created_at":"2026-07-05T11:11:28Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2505.22653/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Recent studies on post-training large language models (LLMs) for reasoning through reinforcement learning (RL) typically focus on tasks that can be accurately verified and rewarded, such as solving math problems. In contrast, our research investigates the impact of reward noise, a more practical consideration for real-world scenarios involving the post-training of LLMs using reward models. We found that LLMs demonstrate strong robustness to substantial reward noise. For example, manually flipping 40% of the reward function's outputs in math tasks still allows a Qwen-2.5-7B model to achieve rap","authors_text":"Ang Lv, Rui Yan, Ruobing Xie, Xingwu Sun, Zhanhui Kang","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-28T17:59:03Z","title":"The Climb Carves Wisdom Deeper Than the Summit: On the Noisy Rewards in Learning to Reason"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.22653","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:c470d52234ad8ec62d0bf33f0d2bf4763da7839d99b6513d7f0f009b19eb26a9","target":"record","created_at":"2026-07-05T11:11:28Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"69b798fffc7b00fd2418675618aa5bd9dbba0e934d2b7347c0572aefb61b7e45","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-28T17:59:03Z","title_canon_sha256":"a38b8656c97931ad1c88b8abf5a587952a0d2f546fa50a71a6e9e5f5f35537de"},"schema_version":"1.0","source":{"id":"2505.22653","kind":"arxiv","version":1}},"canonical_sha256":"143a125fb5e35710eeea5719a7579fac4fc8bb2f3b5fd1a799802e50cf42e3dd","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"143a125fb5e35710eeea5719a7579fac4fc8bb2f3b5fd1a799802e50cf42e3dd","first_computed_at":"2026-07-05T11:11:28.464675Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:11:28.464675Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"MDOV2QYCBhR2DZJ9CtSN21/NZu4fyNgWNRqhWT70UJjsIPzE7zSXyvLEob74E8y9780+T1TX6AzZlQFH5K3GBQ==","signature_status":"signed_v1","signed_at":"2026-07-05T11:11:28.465195Z","signed_message":"canonical_sha256_bytes"},"source_id":"2505.22653","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:c470d52234ad8ec62d0bf33f0d2bf4763da7839d99b6513d7f0f009b19eb26a9","sha256:c95c1d84afbdedd5c78940d00abd589df807d1579fabd0668c37b10a745164e1"],"state_sha256":"508a093c2af304e00889d192a53bd38020804259aff97a19302b1606606e076d"}