{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:MKZNU3KLLEHJF3KRWGZLWSLSFX","short_pith_number":"pith:MKZNU3KL","schema_version":"1.0","canonical_sha256":"62b2da6d4b590e92ed51b1b2bb49722dc9196a95c500c467b378c46654d4a3fe","source":{"kind":"arxiv","id":"2410.15115","version":3},"attestation_state":"computed","paper":{"title":"On Designing Effective RL Reward at Training Time for LLM Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Chuyi He, Guangju Wang, Jiaxuan Gao, Shusheng Xu, Wei Fu, Weilin Liu, Wenjie Ye, Yi Wu, Zhiyu Mei","submitted_at":"2024-10-19T13:53:50Z","abstract_excerpt":"Reward models have been increasingly critical for improving the reasoning capability of LLMs. Existing research has shown that a well-trained reward model can substantially improve model performances at inference time via search. However, the potential of reward models during RL training time still remains largely under-explored. It is currently unclear whether these reward models can provide additional training signals to enhance the reasoning capabilities of LLMs in RL training that uses sparse success rewards, which verify the correctness of solutions. In this work, we evaluate popular rewa"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.15115","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-10-19T13:53:50Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"5be77f22bbfdec265bd581a0d46e5b8fa6b7c2d0e7ca04e105d7a195a1a0552c","abstract_canon_sha256":"4ae72d5291c9371d8cfbb19217c2429e7f14dc784589c3d2b79e6aa2a1508ab9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:41:02.879015Z","signature_b64":"qVvwCnsdtYCUQhIKWvKHfvTaGyTt22oo5xIjyX6/8ytGYbSqTJf9ef3uV/MKlvs+WPyVxTlJxmXA3V6KQFxHDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"62b2da6d4b590e92ed51b1b2bb49722dc9196a95c500c467b378c46654d4a3fe","last_reissued_at":"2026-07-05T09:41:02.878549Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:41:02.878549Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On Designing Effective RL Reward at Training Time for LLM Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Chuyi He, Guangju Wang, Jiaxuan Gao, Shusheng Xu, Wei Fu, Weilin Liu, Wenjie Ye, Yi Wu, Zhiyu Mei","submitted_at":"2024-10-19T13:53:50Z","abstract_excerpt":"Reward models have been increasingly critical for improving the reasoning capability of LLMs. Existing research has shown that a well-trained reward model can substantially improve model performances at inference time via search. However, the potential of reward models during RL training time still remains largely under-explored. It is currently unclear whether these reward models can provide additional training signals to enhance the reasoning capabilities of LLMs in RL training that uses sparse success rewards, which verify the correctness of solutions. In this work, we evaluate popular rewa"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.15115","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.15115/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.15115","created_at":"2026-07-05T09:41:02.878606+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.15115v3","created_at":"2026-07-05T09:41:02.878606+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.15115","created_at":"2026-07-05T09:41:02.878606+00:00"},{"alias_kind":"pith_short_12","alias_value":"MKZNU3KLLEHJ","created_at":"2026-07-05T09:41:02.878606+00:00"},{"alias_kind":"pith_short_16","alias_value":"MKZNU3KLLEHJF3KR","created_at":"2026-07-05T09:41:02.878606+00:00"},{"alias_kind":"pith_short_8","alias_value":"MKZNU3KL","created_at":"2026-07-05T09:41:02.878606+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":16,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25832","citing_title":"MiniOpt: Reasoning to Model and Solve General Optimization Problems with Limited Resources","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2606.25832","citing_title":"MiniOpt: Reasoning to Model and Solve General Optimization Problems with Limited Resources","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18810","citing_title":"Learning from Own Solutions: Self-Conditioned Credit Assignment for Reinforcement Learning with Verifiable Rewards","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17803","citing_title":"Continual Self-Improvement with Lightweight Experiential Latent Memories","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03073","citing_title":"Efficient Hyperparameter Optimization for LLM Reinforcement Learning","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01075","citing_title":"On the Generalization Gap in Self-Evolving Language Model Reasoning","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2511.00066","citing_title":"Sharpness-Guided Group Relative Policy Optimization via Probability Shaping","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2602.03452","citing_title":"Beyond Variance: Prompt-Efficient RLVR via Rare-Event Amplification and Bidirectional Pairing","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2504.20571","citing_title":"Reinforcement Learning for Reasoning in Large Language Models with One Training Example","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2603.19880","citing_title":"What If Consensus Lies? Selective-Complementary Reinforcement Learning at Test Time","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2507.21046","citing_title":"A Survey of Self-Evolving Agents: What, When, How, and Where to Evolve on the Path to Artificial Super Intelligence","ref_index":126,"is_internal_anchor":false},{"citing_arxiv_id":"2503.21776","citing_title":"Video-R1: Reinforcing Video Reasoning in MLLMs","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2503.09567","citing_title":"Towards Reasoning Era: A Survey of Long Chain-of-Thought for Reasoning Large Language Models","ref_index":200,"is_internal_anchor":false},{"citing_arxiv_id":"2502.01456","citing_title":"Process Reinforcement through Implicit Rewards","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18966","citing_title":"Self-Improving Tabular Language Models via Iterative Reward-Guided Post-Training","ref_index":222,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00015","citing_title":"TimeRFT: Stimulating Generalizable Time Series Forecasting for TSFMs via Reinforcement Finetuning","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MKZNU3KLLEHJF3KRWGZLWSLSFX","json":"https://pith.science/pith/MKZNU3KLLEHJF3KRWGZLWSLSFX.json","graph_json":"https://pith.science/api/pith-number/MKZNU3KLLEHJF3KRWGZLWSLSFX/graph.json","events_json":"https://pith.science/api/pith-number/MKZNU3KLLEHJF3KRWGZLWSLSFX/events.json","paper":"https://pith.science/paper/MKZNU3KL"},"agent_actions":{"view_html":"https://pith.science/pith/MKZNU3KLLEHJF3KRWGZLWSLSFX","download_json":"https://pith.science/pith/MKZNU3KLLEHJF3KRWGZLWSLSFX.json","view_paper":"https://pith.science/paper/MKZNU3KL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.15115&json=true","fetch_graph":"https://pith.science/api/pith-number/MKZNU3KLLEHJF3KRWGZLWSLSFX/graph.json","fetch_events":"https://pith.science/api/pith-number/MKZNU3KLLEHJF3KRWGZLWSLSFX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MKZNU3KLLEHJF3KRWGZLWSLSFX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MKZNU3KLLEHJF3KRWGZLWSLSFX/action/storage_attestation","attest_author":"https://pith.science/pith/MKZNU3KLLEHJF3KRWGZLWSLSFX/action/author_attestation","sign_citation":"https://pith.science/pith/MKZNU3KLLEHJF3KRWGZLWSLSFX/action/citation_signature","submit_replication":"https://pith.science/pith/MKZNU3KLLEHJF3KRWGZLWSLSFX/action/replication_record"}},"created_at":"2026-07-05T09:41:02.878606+00:00","updated_at":"2026-07-05T09:41:02.878606+00:00"}