{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:HBWLYSAPEZHGDNRY5WNTH752J4","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"9b233325de9f0d2356eebdf2a9dd81a62c4b06edd6dde0f9d4de2bf6afa61462","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-02-16T20:00:56Z","title_canon_sha256":"17df1bbc33cfcbac26c931f01c1fbb3e3e625d0b1d7fa9ad1050d323684150c8"},"schema_version":"1.0","source":{"id":"2502.11250","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2502.11250","created_at":"2026-07-05T10:15:05Z"},{"alias_kind":"arxiv_version","alias_value":"2502.11250v1","created_at":"2026-07-05T10:15:05Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.11250","created_at":"2026-07-05T10:15:05Z"},{"alias_kind":"pith_short_12","alias_value":"HBWLYSAPEZHG","created_at":"2026-07-05T10:15:05Z"},{"alias_kind":"pith_short_16","alias_value":"HBWLYSAPEZHGDNRY","created_at":"2026-07-05T10:15:05Z"},{"alias_kind":"pith_short_8","alias_value":"HBWLYSAP","created_at":"2026-07-05T10:15:05Z"}],"graph_snapshots":[{"event_id":"sha256:b40399876003789224d36eac832976243a92fc62e6a0718600e0cff18bea2753","target":"graph","created_at":"2026-07-05T10:15:05Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2502.11250/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Complex multi-step reasoning tasks, such as solving mathematical problems, remain challenging for large language models (LLMs). While outcome supervision is commonly used, process supervision via process reward models (PRMs) provides intermediate rewards to verify step-wise correctness in solution traces. However, as proxies for human judgement, PRMs suffer from reliability issues, including susceptibility to reward hacking. In this work, we propose leveraging uncertainty quantification (UQ) to enhance the reliability of step-wise verification with generative reward models for mathematical rea","authors_text":"Luckeciano Carvalho Melo, Phil Blunsom, Sam Staton, Yarin Gal, Younesse Kaddar, Zihuiwen Ye","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-02-16T20:00:56Z","title":"Uncertainty-Aware Step-wise Verification with Generative Reward Models"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.11250","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:32d4eed73ebcc139745709b809996839834ec4cb10e3f73a1d18b9a707a81ebc","target":"record","created_at":"2026-07-05T10:15:05Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"9b233325de9f0d2356eebdf2a9dd81a62c4b06edd6dde0f9d4de2bf6afa61462","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-02-16T20:00:56Z","title_canon_sha256":"17df1bbc33cfcbac26c931f01c1fbb3e3e625d0b1d7fa9ad1050d323684150c8"},"schema_version":"1.0","source":{"id":"2502.11250","kind":"arxiv","version":1}},"canonical_sha256":"386cbc480f264e61b638ed9b33ffba4f2c95721b49f19e108a766e8324af7f10","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"386cbc480f264e61b638ed9b33ffba4f2c95721b49f19e108a766e8324af7f10","first_computed_at":"2026-07-05T10:15:05.701336Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T10:15:05.701336Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"wbFR9f1a4iNbWPa8Z739oKywYHPUoDeANQHk0nq5GldKehyGjcJKuPGfIjFUfdMzn9URczAaBvbHrpjdBrnADg==","signature_status":"signed_v1","signed_at":"2026-07-05T10:15:05.701824Z","signed_message":"canonical_sha256_bytes"},"source_id":"2502.11250","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:32d4eed73ebcc139745709b809996839834ec4cb10e3f73a1d18b9a707a81ebc","sha256:b40399876003789224d36eac832976243a92fc62e6a0718600e0cff18bea2753"],"state_sha256":"5d0d547c96bf8511a022c8dd5acc9c3a19aa0eab052b8df9f5345e669ab21b00"}