{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:MKZNU3KLLEHJF3KRWGZLWSLSFX","short_pith_number":"pith:MKZNU3KL","canonical_record":{"source":{"id":"2410.15115","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-10-19T13:53:50Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"5be77f22bbfdec265bd581a0d46e5b8fa6b7c2d0e7ca04e105d7a195a1a0552c","abstract_canon_sha256":"4ae72d5291c9371d8cfbb19217c2429e7f14dc784589c3d2b79e6aa2a1508ab9"},"schema_version":"1.0"},"canonical_sha256":"62b2da6d4b590e92ed51b1b2bb49722dc9196a95c500c467b378c46654d4a3fe","source":{"kind":"arxiv","id":"2410.15115","version":3},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2410.15115","created_at":"2026-07-05T09:41:02Z"},{"alias_kind":"arxiv_version","alias_value":"2410.15115v3","created_at":"2026-07-05T09:41:02Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.15115","created_at":"2026-07-05T09:41:02Z"},{"alias_kind":"pith_short_12","alias_value":"MKZNU3KLLEHJ","created_at":"2026-07-05T09:41:02Z"},{"alias_kind":"pith_short_16","alias_value":"MKZNU3KLLEHJF3KR","created_at":"2026-07-05T09:41:02Z"},{"alias_kind":"pith_short_8","alias_value":"MKZNU3KL","created_at":"2026-07-05T09:41:02Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:MKZNU3KLLEHJF3KRWGZLWSLSFX","target":"record","payload":{"canonical_record":{"source":{"id":"2410.15115","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-10-19T13:53:50Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"5be77f22bbfdec265bd581a0d46e5b8fa6b7c2d0e7ca04e105d7a195a1a0552c","abstract_canon_sha256":"4ae72d5291c9371d8cfbb19217c2429e7f14dc784589c3d2b79e6aa2a1508ab9"},"schema_version":"1.0"},"canonical_sha256":"62b2da6d4b590e92ed51b1b2bb49722dc9196a95c500c467b378c46654d4a3fe","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:41:02.879015Z","signature_b64":"qVvwCnsdtYCUQhIKWvKHfvTaGyTt22oo5xIjyX6/8ytGYbSqTJf9ef3uV/MKlvs+WPyVxTlJxmXA3V6KQFxHDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"62b2da6d4b590e92ed51b1b2bb49722dc9196a95c500c467b378c46654d4a3fe","last_reissued_at":"2026-07-05T09:41:02.878549Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:41:02.878549Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2410.15115","source_version":3,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:41:02Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"zBYqvwNhclCyUAJVO87JYNl8QKC02rN9ZnjZaao7uL3AxxGRRsUgExSM9rfzGHazjGwHbh7pQLPRD/JLDk1EDw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-07-20T17:48:21.158486Z"},"content_sha256":"5f8ac49483321244c9e64faa99d9fca70534aa99b65316e5d4a8bb03bac83627","schema_version":"1.0","event_id":"sha256:5f8ac49483321244c9e64faa99d9fca70534aa99b65316e5d4a8bb03bac83627"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:MKZNU3KLLEHJF3KRWGZLWSLSFX","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"On Designing Effective RL Reward at Training Time for LLM Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Chuyi He, Guangju Wang, Jiaxuan Gao, Shusheng Xu, Wei Fu, Weilin Liu, Wenjie Ye, Yi Wu, Zhiyu Mei","submitted_at":"2024-10-19T13:53:50Z","abstract_excerpt":"Reward models have been increasingly critical for improving the reasoning capability of LLMs. Existing research has shown that a well-trained reward model can substantially improve model performances at inference time via search. However, the potential of reward models during RL training time still remains largely under-explored. It is currently unclear whether these reward models can provide additional training signals to enhance the reasoning capabilities of LLMs in RL training that uses sparse success rewards, which verify the correctness of solutions. In this work, we evaluate popular rewa"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.15115","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.15115/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:41:02Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"jjJI/okW8UqcKkygeZ2GR2L3CSMHpNBBCZR2hej9lhusOWJt3QhPTVBITzCO+khV7OfAHdTZlvTNCv9eifjoDA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-07-20T17:48:21.158868Z"},"content_sha256":"607c656b07e8b4e1a904d1b90e294a903729dd011da2914f031f58937b47e10b","schema_version":"1.0","event_id":"sha256:607c656b07e8b4e1a904d1b90e294a903729dd011da2914f031f58937b47e10b"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/MKZNU3KLLEHJF3KRWGZLWSLSFX/bundle.json","state_url":"https://pith.science/pith/MKZNU3KLLEHJF3KRWGZLWSLSFX/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/MKZNU3KLLEHJF3KRWGZLWSLSFX/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-07-20T17:48:21Z","links":{"resolver":"https://pith.science/pith/MKZNU3KLLEHJF3KRWGZLWSLSFX","bundle":"https://pith.science/pith/MKZNU3KLLEHJF3KRWGZLWSLSFX/bundle.json","state":"https://pith.science/pith/MKZNU3KLLEHJF3KRWGZLWSLSFX/state.json","well_known_bundle":"https://pith.science/.well-known/pith/MKZNU3KLLEHJF3KRWGZLWSLSFX/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:MKZNU3KLLEHJF3KRWGZLWSLSFX","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"4ae72d5291c9371d8cfbb19217c2429e7f14dc784589c3d2b79e6aa2a1508ab9","cross_cats_sorted":["cs.AI","cs.CL"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-10-19T13:53:50Z","title_canon_sha256":"5be77f22bbfdec265bd581a0d46e5b8fa6b7c2d0e7ca04e105d7a195a1a0552c"},"schema_version":"1.0","source":{"id":"2410.15115","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2410.15115","created_at":"2026-07-05T09:41:02Z"},{"alias_kind":"arxiv_version","alias_value":"2410.15115v3","created_at":"2026-07-05T09:41:02Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.15115","created_at":"2026-07-05T09:41:02Z"},{"alias_kind":"pith_short_12","alias_value":"MKZNU3KLLEHJ","created_at":"2026-07-05T09:41:02Z"},{"alias_kind":"pith_short_16","alias_value":"MKZNU3KLLEHJF3KR","created_at":"2026-07-05T09:41:02Z"},{"alias_kind":"pith_short_8","alias_value":"MKZNU3KL","created_at":"2026-07-05T09:41:02Z"}],"graph_snapshots":[{"event_id":"sha256:607c656b07e8b4e1a904d1b90e294a903729dd011da2914f031f58937b47e10b","target":"graph","created_at":"2026-07-05T09:41:02Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2410.15115/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reward models have been increasingly critical for improving the reasoning capability of LLMs. Existing research has shown that a well-trained reward model can substantially improve model performances at inference time via search. However, the potential of reward models during RL training time still remains largely under-explored. It is currently unclear whether these reward models can provide additional training signals to enhance the reasoning capabilities of LLMs in RL training that uses sparse success rewards, which verify the correctness of solutions. In this work, we evaluate popular rewa","authors_text":"Chuyi He, Guangju Wang, Jiaxuan Gao, Shusheng Xu, Wei Fu, Weilin Liu, Wenjie Ye, Yi Wu, Zhiyu Mei","cross_cats":["cs.AI","cs.CL"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-10-19T13:53:50Z","title":"On Designing Effective RL Reward at Training Time for LLM Reasoning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.15115","kind":"arxiv","version":3},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:5f8ac49483321244c9e64faa99d9fca70534aa99b65316e5d4a8bb03bac83627","target":"record","created_at":"2026-07-05T09:41:02Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"4ae72d5291c9371d8cfbb19217c2429e7f14dc784589c3d2b79e6aa2a1508ab9","cross_cats_sorted":["cs.AI","cs.CL"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-10-19T13:53:50Z","title_canon_sha256":"5be77f22bbfdec265bd581a0d46e5b8fa6b7c2d0e7ca04e105d7a195a1a0552c"},"schema_version":"1.0","source":{"id":"2410.15115","kind":"arxiv","version":3}},"canonical_sha256":"62b2da6d4b590e92ed51b1b2bb49722dc9196a95c500c467b378c46654d4a3fe","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"62b2da6d4b590e92ed51b1b2bb49722dc9196a95c500c467b378c46654d4a3fe","first_computed_at":"2026-07-05T09:41:02.878549Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T09:41:02.878549Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"qVvwCnsdtYCUQhIKWvKHfvTaGyTt22oo5xIjyX6/8ytGYbSqTJf9ef3uV/MKlvs+WPyVxTlJxmXA3V6KQFxHDg==","signature_status":"signed_v1","signed_at":"2026-07-05T09:41:02.879015Z","signed_message":"canonical_sha256_bytes"},"source_id":"2410.15115","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:5f8ac49483321244c9e64faa99d9fca70534aa99b65316e5d4a8bb03bac83627","sha256:607c656b07e8b4e1a904d1b90e294a903729dd011da2914f031f58937b47e10b"],"state_sha256":"f95f3548680d727a3a549a303ffaf4ca3fc88182a29fb8b804a5866f713c8876"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"dzulfUKwiSbgXEkMzVQMmDwIbagXY9PMQwuBtuezei1iGqkaM9jtq/mE8dGFW8qSaj0Ew6d3REp+57et/LhsCw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-07-20T17:48:21.161041Z","bundle_sha256":"0ee604bccb2059c967c8566a31e2faa380724e681ebb18316abb1c65d1fdd9f7"}}