{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:HBWLYSAPEZHGDNRY5WNTH752J4","short_pith_number":"pith:HBWLYSAP","schema_version":"1.0","canonical_sha256":"386cbc480f264e61b638ed9b33ffba4f2c95721b49f19e108a766e8324af7f10","source":{"kind":"arxiv","id":"2502.11250","version":1},"attestation_state":"computed","paper":{"title":"Uncertainty-Aware Step-wise Verification with Generative Reward Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Luckeciano Carvalho Melo, Phil Blunsom, Sam Staton, Yarin Gal, Younesse Kaddar, Zihuiwen Ye","submitted_at":"2025-02-16T20:00:56Z","abstract_excerpt":"Complex multi-step reasoning tasks, such as solving mathematical problems, remain challenging for large language models (LLMs). While outcome supervision is commonly used, process supervision via process reward models (PRMs) provides intermediate rewards to verify step-wise correctness in solution traces. However, as proxies for human judgement, PRMs suffer from reliability issues, including susceptibility to reward hacking. In this work, we propose leveraging uncertainty quantification (UQ) to enhance the reliability of step-wise verification with generative reward models for mathematical rea"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.11250","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-02-16T20:00:56Z","cross_cats_sorted":[],"title_canon_sha256":"17df1bbc33cfcbac26c931f01c1fbb3e3e625d0b1d7fa9ad1050d323684150c8","abstract_canon_sha256":"9b233325de9f0d2356eebdf2a9dd81a62c4b06edd6dde0f9d4de2bf6afa61462"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:15:05.701824Z","signature_b64":"wbFR9f1a4iNbWPa8Z739oKywYHPUoDeANQHk0nq5GldKehyGjcJKuPGfIjFUfdMzn9URczAaBvbHrpjdBrnADg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"386cbc480f264e61b638ed9b33ffba4f2c95721b49f19e108a766e8324af7f10","last_reissued_at":"2026-07-05T10:15:05.701336Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:15:05.701336Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Uncertainty-Aware Step-wise Verification with Generative Reward Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Luckeciano Carvalho Melo, Phil Blunsom, Sam Staton, Yarin Gal, Younesse Kaddar, Zihuiwen Ye","submitted_at":"2025-02-16T20:00:56Z","abstract_excerpt":"Complex multi-step reasoning tasks, such as solving mathematical problems, remain challenging for large language models (LLMs). While outcome supervision is commonly used, process supervision via process reward models (PRMs) provides intermediate rewards to verify step-wise correctness in solution traces. However, as proxies for human judgement, PRMs suffer from reliability issues, including susceptibility to reward hacking. In this work, we propose leveraging uncertainty quantification (UQ) to enhance the reliability of step-wise verification with generative reward models for mathematical rea"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.11250","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.11250/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.11250","created_at":"2026-07-05T10:15:05.701396+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.11250v1","created_at":"2026-07-05T10:15:05.701396+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.11250","created_at":"2026-07-05T10:15:05.701396+00:00"},{"alias_kind":"pith_short_12","alias_value":"HBWLYSAPEZHG","created_at":"2026-07-05T10:15:05.701396+00:00"},{"alias_kind":"pith_short_16","alias_value":"HBWLYSAPEZHGDNRY","created_at":"2026-07-05T10:15:05.701396+00:00"},{"alias_kind":"pith_short_8","alias_value":"HBWLYSAP","created_at":"2026-07-05T10:15:05.701396+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22792","citing_title":"The Origins of Stochasticity: Comprehensive Investigations on Uncertainty Quantification for Large Language Models","ref_index":119,"is_internal_anchor":false},{"citing_arxiv_id":"2505.11737","citing_title":"TokUR: Token-Level Uncertainty Estimation for Large Language Model Reasoning","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21602","citing_title":"Benchmarking and Improving Monitors for Out-Of-Distribution Alignment Failure in LLMs","ref_index":43,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HBWLYSAPEZHGDNRY5WNTH752J4","json":"https://pith.science/pith/HBWLYSAPEZHGDNRY5WNTH752J4.json","graph_json":"https://pith.science/api/pith-number/HBWLYSAPEZHGDNRY5WNTH752J4/graph.json","events_json":"https://pith.science/api/pith-number/HBWLYSAPEZHGDNRY5WNTH752J4/events.json","paper":"https://pith.science/paper/HBWLYSAP"},"agent_actions":{"view_html":"https://pith.science/pith/HBWLYSAPEZHGDNRY5WNTH752J4","download_json":"https://pith.science/pith/HBWLYSAPEZHGDNRY5WNTH752J4.json","view_paper":"https://pith.science/paper/HBWLYSAP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.11250&json=true","fetch_graph":"https://pith.science/api/pith-number/HBWLYSAPEZHGDNRY5WNTH752J4/graph.json","fetch_events":"https://pith.science/api/pith-number/HBWLYSAPEZHGDNRY5WNTH752J4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HBWLYSAPEZHGDNRY5WNTH752J4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HBWLYSAPEZHGDNRY5WNTH752J4/action/storage_attestation","attest_author":"https://pith.science/pith/HBWLYSAPEZHGDNRY5WNTH752J4/action/author_attestation","sign_citation":"https://pith.science/pith/HBWLYSAPEZHGDNRY5WNTH752J4/action/citation_signature","submit_replication":"https://pith.science/pith/HBWLYSAPEZHGDNRY5WNTH752J4/action/replication_record"}},"created_at":"2026-07-05T10:15:05.701396+00:00","updated_at":"2026-07-05T10:15:05.701396+00:00"}