{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2023:6P7VNA25W4FGIAJV2NASYNRHGM","short_pith_number":"pith:6P7VNA25","canonical_record":{"source":{"id":"2310.10080","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2023-10-16T05:21:50Z","cross_cats_sorted":[],"title_canon_sha256":"15272afc4334d72abe4aabe7ebc8149d0b135c8ccb248fb68afb191bd06aa101","abstract_canon_sha256":"7fefbbf505a6d483f28f1ed988857e172d4bd9d67f68da505ccd1976ca925b9e"},"schema_version":"1.0"},"canonical_sha256":"f3ff56835db70a640135d3412c36273331a11b68f546ccf5574162fa9b8c1fe2","source":{"kind":"arxiv","id":"2310.10080","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2310.10080","created_at":"2026-07-05T07:01:14Z"},{"alias_kind":"arxiv_version","alias_value":"2310.10080v1","created_at":"2026-07-05T07:01:14Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.10080","created_at":"2026-07-05T07:01:14Z"},{"alias_kind":"pith_short_12","alias_value":"6P7VNA25W4FG","created_at":"2026-07-05T07:01:14Z"},{"alias_kind":"pith_short_16","alias_value":"6P7VNA25W4FGIAJV","created_at":"2026-07-05T07:01:14Z"},{"alias_kind":"pith_short_8","alias_value":"6P7VNA25","created_at":"2026-07-05T07:01:14Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2023:6P7VNA25W4FGIAJV2NASYNRHGM","target":"record","payload":{"canonical_record":{"source":{"id":"2310.10080","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2023-10-16T05:21:50Z","cross_cats_sorted":[],"title_canon_sha256":"15272afc4334d72abe4aabe7ebc8149d0b135c8ccb248fb68afb191bd06aa101","abstract_canon_sha256":"7fefbbf505a6d483f28f1ed988857e172d4bd9d67f68da505ccd1976ca925b9e"},"schema_version":"1.0"},"canonical_sha256":"f3ff56835db70a640135d3412c36273331a11b68f546ccf5574162fa9b8c1fe2","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:01:14.992217Z","signature_b64":"oFt0ZHr2FVBcS3T/F+XNf6smZkzB7nM6i9uG6RU3RlCQAU/xXGfnBTpT8RSKr6fIhgkj3OUI4P+yzcjcFs+PBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f3ff56835db70a640135d3412c36273331a11b68f546ccf5574162fa9b8c1fe2","last_reissued_at":"2026-07-05T07:01:14.991750Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:01:14.991750Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2310.10080","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T07:01:14Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"OQJ5O/VI3hkX0thUQGDt54hF45r3ucoReiwgiSpclTSnQq5HDpFZaq+N52fXTiw+i7pXtCJsXr+yTPz7ASklAg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-14T09:25:26.260692Z"},"content_sha256":"ece17b62614904a7fe5cca596bf7605404e6500fe7540b526790b527598ea0ff","schema_version":"1.0","event_id":"sha256:ece17b62614904a7fe5cca596bf7605404e6500fe7540b526790b527598ea0ff"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2023:6P7VNA25W4FGIAJV2NASYNRHGM","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Let's reward step by step: Step-Level reward model as the Navigators for Reasoning","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Haotian Zhou, Hongxia Yang, Jianbo Yuan, Pengfei Liu, Qianli Ma, Tingkai Liu, Yang You","submitted_at":"2023-10-16T05:21:50Z","abstract_excerpt":"Recent years have seen considerable advancements in multi-step reasoning with Large Language Models (LLMs). The previous studies have elucidated the merits of integrating feedback or search mechanisms during model inference to improve the reasoning accuracy. The Process-Supervised Reward Model (PRM), typically furnishes LLMs with step-by-step feedback during the training phase, akin to Proximal Policy Optimization (PPO) or reject sampling. Our objective is to examine the efficacy of PRM in the inference phase to help discern the optimal solution paths for multi-step tasks such as mathematical "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.10080","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.10080/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T07:01:14Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"pNz1g0IfIaFN/tLYOmS08ni0H1OGAEXNWchRK1BkVCLhJqEDfdR2gb9/rsX4jJ+orK0CgGUJPR5r6VcKDit+Cw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-14T09:25:26.261034Z"},"content_sha256":"2069fbbe73518b7b588d022a54b12ac05808170a9a1c9c9ed2eafdd35b3a6db7","schema_version":"1.0","event_id":"sha256:2069fbbe73518b7b588d022a54b12ac05808170a9a1c9c9ed2eafdd35b3a6db7"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/6P7VNA25W4FGIAJV2NASYNRHGM/bundle.json","state_url":"https://pith.science/pith/6P7VNA25W4FGIAJV2NASYNRHGM/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/6P7VNA25W4FGIAJV2NASYNRHGM/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-14T09:25:26Z","links":{"resolver":"https://pith.science/pith/6P7VNA25W4FGIAJV2NASYNRHGM","bundle":"https://pith.science/pith/6P7VNA25W4FGIAJV2NASYNRHGM/bundle.json","state":"https://pith.science/pith/6P7VNA25W4FGIAJV2NASYNRHGM/state.json","well_known_bundle":"https://pith.science/.well-known/pith/6P7VNA25W4FGIAJV2NASYNRHGM/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2023:6P7VNA25W4FGIAJV2NASYNRHGM","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"7fefbbf505a6d483f28f1ed988857e172d4bd9d67f68da505ccd1976ca925b9e","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2023-10-16T05:21:50Z","title_canon_sha256":"15272afc4334d72abe4aabe7ebc8149d0b135c8ccb248fb68afb191bd06aa101"},"schema_version":"1.0","source":{"id":"2310.10080","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2310.10080","created_at":"2026-07-05T07:01:14Z"},{"alias_kind":"arxiv_version","alias_value":"2310.10080v1","created_at":"2026-07-05T07:01:14Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.10080","created_at":"2026-07-05T07:01:14Z"},{"alias_kind":"pith_short_12","alias_value":"6P7VNA25W4FG","created_at":"2026-07-05T07:01:14Z"},{"alias_kind":"pith_short_16","alias_value":"6P7VNA25W4FGIAJV","created_at":"2026-07-05T07:01:14Z"},{"alias_kind":"pith_short_8","alias_value":"6P7VNA25","created_at":"2026-07-05T07:01:14Z"}],"graph_snapshots":[{"event_id":"sha256:2069fbbe73518b7b588d022a54b12ac05808170a9a1c9c9ed2eafdd35b3a6db7","target":"graph","created_at":"2026-07-05T07:01:14Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2310.10080/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Recent years have seen considerable advancements in multi-step reasoning with Large Language Models (LLMs). The previous studies have elucidated the merits of integrating feedback or search mechanisms during model inference to improve the reasoning accuracy. The Process-Supervised Reward Model (PRM), typically furnishes LLMs with step-by-step feedback during the training phase, akin to Proximal Policy Optimization (PPO) or reject sampling. Our objective is to examine the efficacy of PRM in the inference phase to help discern the optimal solution paths for multi-step tasks such as mathematical ","authors_text":"Haotian Zhou, Hongxia Yang, Jianbo Yuan, Pengfei Liu, Qianli Ma, Tingkai Liu, Yang You","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2023-10-16T05:21:50Z","title":"Let's reward step by step: Step-Level reward model as the Navigators for Reasoning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.10080","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:ece17b62614904a7fe5cca596bf7605404e6500fe7540b526790b527598ea0ff","target":"record","created_at":"2026-07-05T07:01:14Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"7fefbbf505a6d483f28f1ed988857e172d4bd9d67f68da505ccd1976ca925b9e","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2023-10-16T05:21:50Z","title_canon_sha256":"15272afc4334d72abe4aabe7ebc8149d0b135c8ccb248fb68afb191bd06aa101"},"schema_version":"1.0","source":{"id":"2310.10080","kind":"arxiv","version":1}},"canonical_sha256":"f3ff56835db70a640135d3412c36273331a11b68f546ccf5574162fa9b8c1fe2","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"f3ff56835db70a640135d3412c36273331a11b68f546ccf5574162fa9b8c1fe2","first_computed_at":"2026-07-05T07:01:14.991750Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T07:01:14.991750Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"oFt0ZHr2FVBcS3T/F+XNf6smZkzB7nM6i9uG6RU3RlCQAU/xXGfnBTpT8RSKr6fIhgkj3OUI4P+yzcjcFs+PBw==","signature_status":"signed_v1","signed_at":"2026-07-05T07:01:14.992217Z","signed_message":"canonical_sha256_bytes"},"source_id":"2310.10080","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:ece17b62614904a7fe5cca596bf7605404e6500fe7540b526790b527598ea0ff","sha256:2069fbbe73518b7b588d022a54b12ac05808170a9a1c9c9ed2eafdd35b3a6db7"],"state_sha256":"b7f9126bb6255852879e0842ea19d59eb7add19209810bf5d52735bbd3f89708"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"asJevnkcH4gs73dPvvzoTFCPOmZI0pk1lfu8kQriGvcaizkyik0mrU9ngResvZNW0mqltOwudZmSGSk/2PwnBA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-14T09:25:26.264889Z","bundle_sha256":"0ab88565fcbeee1de270601d21b2e58d8c738a122f4ea12330c753117eb9aa53"}}