{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:JL23K5C4KYG4AF7K53WJJUAY3U","short_pith_number":"pith:JL23K5C4","schema_version":"1.0","canonical_sha256":"4af5b5745c560dc017eaeeec94d018dd270e2ea8d2e0779d735d86643cd35e50","source":{"kind":"arxiv","id":"2508.11800","version":1},"attestation_state":"computed","paper":{"title":"Uncalibrated Reasoning: GRPO Induces Overconfidence for Stochastic Outcomes","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Jure Leskovec, Michael Bereket","submitted_at":"2025-08-15T20:50:53Z","abstract_excerpt":"Reinforcement learning (RL) has proven remarkably effective at improving the accuracy of language models in verifiable and deterministic domains like mathematics. Here, we examine if current RL methods are also effective at optimizing language models in verifiable domains with stochastic outcomes, like scientific experiments. Through applications to synthetic data and real-world biological experiments, we demonstrate that Group Relative Policy Optimization (GRPO) induces overconfident probability predictions for binary stochastic outcomes, while Proximal Policy Optimization (PPO) and REINFORCE"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.11800","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-08-15T20:50:53Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"83f20d1a2a7dd4a8c42dd62ef166e8c3217539be6ee78c5edff779e19acf8617","abstract_canon_sha256":"c3b47b74bad2ee05b1732e7d303acded8de2e0849be8931b5a445d39d162695a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:54:54.672248Z","signature_b64":"h7RS7vFjCBrRBrhL6POCoFLpjx4OUBQRaRDwRph6njSZPAw3aECn1rXpxYttjkCdpFi5WX+lT2lh0ZKWwPr+CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4af5b5745c560dc017eaeeec94d018dd270e2ea8d2e0779d735d86643cd35e50","last_reissued_at":"2026-07-05T11:54:54.671739Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:54:54.671739Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Uncalibrated Reasoning: GRPO Induces Overconfidence for Stochastic Outcomes","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Jure Leskovec, Michael Bereket","submitted_at":"2025-08-15T20:50:53Z","abstract_excerpt":"Reinforcement learning (RL) has proven remarkably effective at improving the accuracy of language models in verifiable and deterministic domains like mathematics. Here, we examine if current RL methods are also effective at optimizing language models in verifiable domains with stochastic outcomes, like scientific experiments. Through applications to synthetic data and real-world biological experiments, we demonstrate that Group Relative Policy Optimization (GRPO) induces overconfident probability predictions for binary stochastic outcomes, while Proximal Policy Optimization (PPO) and REINFORCE"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.11800","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.11800/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.11800","created_at":"2026-07-05T11:54:54.671805+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.11800v1","created_at":"2026-07-05T11:54:54.671805+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.11800","created_at":"2026-07-05T11:54:54.671805+00:00"},{"alias_kind":"pith_short_12","alias_value":"JL23K5C4KYG4","created_at":"2026-07-05T11:54:54.671805+00:00"},{"alias_kind":"pith_short_16","alias_value":"JL23K5C4KYG4AF7K","created_at":"2026-07-05T11:54:54.671805+00:00"},{"alias_kind":"pith_short_8","alias_value":"JL23K5C4","created_at":"2026-07-05T11:54:54.671805+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.00164","citing_title":"Verifiable Rewards for Calibrated Probabilistic Forecasting","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2509.25454","citing_title":"DeepSearch: Overcome the Bottleneck of Reinforcement Learning with Verifiable Rewards via Monte Carlo Tree Search","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2512.03043","citing_title":"OneThinker: All-in-one Reasoning Model for Image and Video","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2603.09117","citing_title":"Decoupling Reasoning and Confidence: Resurrecting Calibration in Reinforcement Learning from Verifiable Rewards","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10937","citing_title":"Power Reinforcement Post-Training of Text-to-Image Models with Super-Linear Advantage Shaping","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12632","citing_title":"Calibration-Aware Policy Optimization for Reasoning LLMs","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08539","citing_title":"OpenVLThinkerV2: A Generalist Multimodal Reasoning Model for Multi-domain Visual Tasks","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JL23K5C4KYG4AF7K53WJJUAY3U","json":"https://pith.science/pith/JL23K5C4KYG4AF7K53WJJUAY3U.json","graph_json":"https://pith.science/api/pith-number/JL23K5C4KYG4AF7K53WJJUAY3U/graph.json","events_json":"https://pith.science/api/pith-number/JL23K5C4KYG4AF7K53WJJUAY3U/events.json","paper":"https://pith.science/paper/JL23K5C4"},"agent_actions":{"view_html":"https://pith.science/pith/JL23K5C4KYG4AF7K53WJJUAY3U","download_json":"https://pith.science/pith/JL23K5C4KYG4AF7K53WJJUAY3U.json","view_paper":"https://pith.science/paper/JL23K5C4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.11800&json=true","fetch_graph":"https://pith.science/api/pith-number/JL23K5C4KYG4AF7K53WJJUAY3U/graph.json","fetch_events":"https://pith.science/api/pith-number/JL23K5C4KYG4AF7K53WJJUAY3U/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JL23K5C4KYG4AF7K53WJJUAY3U/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JL23K5C4KYG4AF7K53WJJUAY3U/action/storage_attestation","attest_author":"https://pith.science/pith/JL23K5C4KYG4AF7K53WJJUAY3U/action/author_attestation","sign_citation":"https://pith.science/pith/JL23K5C4KYG4AF7K53WJJUAY3U/action/citation_signature","submit_replication":"https://pith.science/pith/JL23K5C4KYG4AF7K53WJJUAY3U/action/replication_record"}},"created_at":"2026-07-05T11:54:54.671805+00:00","updated_at":"2026-07-05T11:54:54.671805+00:00"}