{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:UYBK7IZBGJ7XA3KKYQ6D4SMPW7","short_pith_number":"pith:UYBK7IZB","canonical_record":{"source":{"id":"2508.19576","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-08-27T05:16:03Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"eaa28280e0f733c1cfc14237f31e80200b12c897c623d5ae2f7e8162d720084c","abstract_canon_sha256":"c33dbd5aacb28e40e86df31ff5e6afe46453d4a59c2faaf1766773f034a4873b"},"schema_version":"1.0"},"canonical_sha256":"a602afa321327f706d4ac43c3e498fb7eef8c9db908caea8b5f2352c521adb3c","source":{"kind":"arxiv","id":"2508.19576","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2508.19576","created_at":"2026-07-05T12:06:18Z"},{"alias_kind":"arxiv_version","alias_value":"2508.19576v2","created_at":"2026-07-05T12:06:18Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.19576","created_at":"2026-07-05T12:06:18Z"},{"alias_kind":"pith_short_12","alias_value":"UYBK7IZBGJ7X","created_at":"2026-07-05T12:06:18Z"},{"alias_kind":"pith_short_16","alias_value":"UYBK7IZBGJ7XA3KK","created_at":"2026-07-05T12:06:18Z"},{"alias_kind":"pith_short_8","alias_value":"UYBK7IZB","created_at":"2026-07-05T12:06:18Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:UYBK7IZBGJ7XA3KKYQ6D4SMPW7","target":"record","payload":{"canonical_record":{"source":{"id":"2508.19576","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-08-27T05:16:03Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"eaa28280e0f733c1cfc14237f31e80200b12c897c623d5ae2f7e8162d720084c","abstract_canon_sha256":"c33dbd5aacb28e40e86df31ff5e6afe46453d4a59c2faaf1766773f034a4873b"},"schema_version":"1.0"},"canonical_sha256":"a602afa321327f706d4ac43c3e498fb7eef8c9db908caea8b5f2352c521adb3c","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:06:18.865209Z","signature_b64":"6ALy/koNRhiMFngvg8kAMbLY0/RYl8ftrcO8FKilDxVkNZwxag4hiUlplcnSuJT9COgLgryeLWzU05hH6XLDDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a602afa321327f706d4ac43c3e498fb7eef8c9db908caea8b5f2352c521adb3c","last_reissued_at":"2026-07-05T12:06:18.864708Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:06:18.864708Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2508.19576","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T12:06:18Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"UkTLR0yxLwTrcEJc2P/yR4LUXJY+9I8UE3FHfWQQapSTwRtK+rtY7k5EWMpX+saN8mruPFxHQoKq+AX6Zqp6AQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-20T07:39:23.710509Z"},"content_sha256":"18f89cff3b37411ad53e668d1359f39e65ffc0870c8a5d679a6db7077d9424bc","schema_version":"1.0","event_id":"sha256:18f89cff3b37411ad53e668d1359f39e65ffc0870c8a5d679a6db7077d9424bc"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:UYBK7IZBGJ7XA3KKYQ6D4SMPW7","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"ReST-RL: Achieving Accurate Code Reasoning of LLMs with Optimized Self-Training and Decoding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Dan Zhang, Jie Tang, Sining Zhoubian","submitted_at":"2025-08-27T05:16:03Z","abstract_excerpt":"With respect to improving the reasoning accuracy of LLMs, the representative reinforcement learning (RL) method GRPO faces failure due to insignificant reward variance, while verification methods based on process reward models (PRMs) suffer from difficulties with training data acquisition and verification effectiveness. To tackle these problems, this paper introduces ReST-RL, a unified LLM RL paradigm that significantly improves LLM's code reasoning ability by combining an improved GRPO algorithm with a meticulously designed test time decoding method assisted by a value model (VM). As the firs"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.19576","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.19576/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T12:06:18Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"oO4tU/i9rn1YF5ccVHKTRdS8Hgwi8bKuuJwDspHnjIxXTQaFBSlQb5z5N5hk+tEDofIXhc/IAWMI1cQMPxQrDQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-20T07:39:23.711028Z"},"content_sha256":"7d884c1e62992ddb33cabce5c5c90802e6a20d0468687429cb70b840a4248747","schema_version":"1.0","event_id":"sha256:7d884c1e62992ddb33cabce5c5c90802e6a20d0468687429cb70b840a4248747"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/UYBK7IZBGJ7XA3KKYQ6D4SMPW7/bundle.json","state_url":"https://pith.science/pith/UYBK7IZBGJ7XA3KKYQ6D4SMPW7/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/UYBK7IZBGJ7XA3KKYQ6D4SMPW7/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-20T07:39:23Z","links":{"resolver":"https://pith.science/pith/UYBK7IZBGJ7XA3KKYQ6D4SMPW7","bundle":"https://pith.science/pith/UYBK7IZBGJ7XA3KKYQ6D4SMPW7/bundle.json","state":"https://pith.science/pith/UYBK7IZBGJ7XA3KKYQ6D4SMPW7/state.json","well_known_bundle":"https://pith.science/.well-known/pith/UYBK7IZBGJ7XA3KKYQ6D4SMPW7/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:UYBK7IZBGJ7XA3KKYQ6D4SMPW7","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"c33dbd5aacb28e40e86df31ff5e6afe46453d4a59c2faaf1766773f034a4873b","cross_cats_sorted":["cs.LG"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-08-27T05:16:03Z","title_canon_sha256":"eaa28280e0f733c1cfc14237f31e80200b12c897c623d5ae2f7e8162d720084c"},"schema_version":"1.0","source":{"id":"2508.19576","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2508.19576","created_at":"2026-07-05T12:06:18Z"},{"alias_kind":"arxiv_version","alias_value":"2508.19576v2","created_at":"2026-07-05T12:06:18Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.19576","created_at":"2026-07-05T12:06:18Z"},{"alias_kind":"pith_short_12","alias_value":"UYBK7IZBGJ7X","created_at":"2026-07-05T12:06:18Z"},{"alias_kind":"pith_short_16","alias_value":"UYBK7IZBGJ7XA3KK","created_at":"2026-07-05T12:06:18Z"},{"alias_kind":"pith_short_8","alias_value":"UYBK7IZB","created_at":"2026-07-05T12:06:18Z"}],"graph_snapshots":[{"event_id":"sha256:7d884c1e62992ddb33cabce5c5c90802e6a20d0468687429cb70b840a4248747","target":"graph","created_at":"2026-07-05T12:06:18Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2508.19576/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"With respect to improving the reasoning accuracy of LLMs, the representative reinforcement learning (RL) method GRPO faces failure due to insignificant reward variance, while verification methods based on process reward models (PRMs) suffer from difficulties with training data acquisition and verification effectiveness. To tackle these problems, this paper introduces ReST-RL, a unified LLM RL paradigm that significantly improves LLM's code reasoning ability by combining an improved GRPO algorithm with a meticulously designed test time decoding method assisted by a value model (VM). As the firs","authors_text":"Dan Zhang, Jie Tang, Sining Zhoubian","cross_cats":["cs.LG"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-08-27T05:16:03Z","title":"ReST-RL: Achieving Accurate Code Reasoning of LLMs with Optimized Self-Training and Decoding"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.19576","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:18f89cff3b37411ad53e668d1359f39e65ffc0870c8a5d679a6db7077d9424bc","target":"record","created_at":"2026-07-05T12:06:18Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"c33dbd5aacb28e40e86df31ff5e6afe46453d4a59c2faaf1766773f034a4873b","cross_cats_sorted":["cs.LG"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-08-27T05:16:03Z","title_canon_sha256":"eaa28280e0f733c1cfc14237f31e80200b12c897c623d5ae2f7e8162d720084c"},"schema_version":"1.0","source":{"id":"2508.19576","kind":"arxiv","version":2}},"canonical_sha256":"a602afa321327f706d4ac43c3e498fb7eef8c9db908caea8b5f2352c521adb3c","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"a602afa321327f706d4ac43c3e498fb7eef8c9db908caea8b5f2352c521adb3c","first_computed_at":"2026-07-05T12:06:18.864708Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T12:06:18.864708Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"6ALy/koNRhiMFngvg8kAMbLY0/RYl8ftrcO8FKilDxVkNZwxag4hiUlplcnSuJT9COgLgryeLWzU05hH6XLDDQ==","signature_status":"signed_v1","signed_at":"2026-07-05T12:06:18.865209Z","signed_message":"canonical_sha256_bytes"},"source_id":"2508.19576","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:18f89cff3b37411ad53e668d1359f39e65ffc0870c8a5d679a6db7077d9424bc","sha256:7d884c1e62992ddb33cabce5c5c90802e6a20d0468687429cb70b840a4248747"],"state_sha256":"4ba1fb7b9f10bfca1c6c311426de760748dfcb98d043f1855b8b295230a01ea1"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Exw5BK2JR4ts2znGFaZnVWPCnn0btjX9RaLtARM3id6vcwY1rqFohAcyiF750iBrtNkfCQuOeShcM7fzVdxKCQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-20T07:39:23.714600Z","bundle_sha256":"a946542c43f5b3f798493f58ddb2384a24d458b47b3ff76e21c237dc5066afaa"}}