{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:RZQ36VSC34G52VNLTYNJLNG32I","short_pith_number":"pith:RZQ36VSC","schema_version":"1.0","canonical_sha256":"8e61bf5642df0ddd55ab9e1a95b4dbd22415f4f4489e7fc7e9d2174778176abe","source":{"kind":"arxiv","id":"2403.04642","version":1},"attestation_state":"computed","paper":{"title":"Teaching Large Language Models to Reason with Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Alex Havrilla, Christoforos Nalmpantis, Eric Hambro, Jane Dwivedi-Yu, Maksym Zhuravinskyi, Roberta Raileanu, Sainbayar Sukhbaatar, Sharath Chandra Raparthy, Yuqing Du","submitted_at":"2024-03-07T16:36:29Z","abstract_excerpt":"Reinforcement Learning from Human Feedback (\\textbf{RLHF}) has emerged as a dominant approach for aligning LLM outputs with human preferences. Inspired by the success of RLHF, we study the performance of multiple algorithms that learn from feedback (Expert Iteration, Proximal Policy Optimization (\\textbf{PPO}), Return-Conditioned RL) on improving LLM reasoning capabilities. We investigate both sparse and dense rewards provided to the LLM both heuristically and via a learned reward model. We additionally start from multiple model sizes and initializations both with and without supervised fine-t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.04642","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-03-07T16:36:29Z","cross_cats_sorted":[],"title_canon_sha256":"73c4c5cfcd5c54623f45f15b76b8fb3a370c2b34f1bbe79dce4c9d0f497ce09b","abstract_canon_sha256":"771da548d567f011e47faff57b551718b8ee3e6a14de9cbb11056c48601aa8de"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:53:26.493923Z","signature_b64":"aMchidx8uueVza6Qdq79pK2/Li53+FbRXViL0JjxPkAU34RzjgF8XIYAfP0tyQ72+8yvZmc2qhvNI0Nef0WjDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8e61bf5642df0ddd55ab9e1a95b4dbd22415f4f4489e7fc7e9d2174778176abe","last_reissued_at":"2026-07-05T07:53:26.493434Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:53:26.493434Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Teaching Large Language Models to Reason with Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Alex Havrilla, Christoforos Nalmpantis, Eric Hambro, Jane Dwivedi-Yu, Maksym Zhuravinskyi, Roberta Raileanu, Sainbayar Sukhbaatar, Sharath Chandra Raparthy, Yuqing Du","submitted_at":"2024-03-07T16:36:29Z","abstract_excerpt":"Reinforcement Learning from Human Feedback (\\textbf{RLHF}) has emerged as a dominant approach for aligning LLM outputs with human preferences. Inspired by the success of RLHF, we study the performance of multiple algorithms that learn from feedback (Expert Iteration, Proximal Policy Optimization (\\textbf{PPO}), Return-Conditioned RL) on improving LLM reasoning capabilities. We investigate both sparse and dense rewards provided to the LLM both heuristically and via a learned reward model. We additionally start from multiple model sizes and initializations both with and without supervised fine-t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.04642","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.04642/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.04642","created_at":"2026-07-05T07:53:26.493495+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.04642v1","created_at":"2026-07-05T07:53:26.493495+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.04642","created_at":"2026-07-05T07:53:26.493495+00:00"},{"alias_kind":"pith_short_12","alias_value":"RZQ36VSC34G5","created_at":"2026-07-05T07:53:26.493495+00:00"},{"alias_kind":"pith_short_16","alias_value":"RZQ36VSC34G52VNL","created_at":"2026-07-05T07:53:26.493495+00:00"},{"alias_kind":"pith_short_8","alias_value":"RZQ36VSC","created_at":"2026-07-05T07:53:26.493495+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":17,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2502.12272","citing_title":"Learning to Reason at the Frontier of Learnability","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2410.08146","citing_title":"Rewarding Progress: Scaling Automated Process Verifiers for LLM Reasoning","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2505.23912","citing_title":"LoVeC: Reinforcement Learning for Better Verbalized Confidence in Long-Form Generations","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2508.00222","citing_title":"RL-PLUS: Countering Capability Boundary Collapse of LLMs in Reinforcement Learning with Hybrid-policy Optimization","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2508.16745","citing_title":"Beyond Memorization: Extending Reasoning Depth with Recurrence, Memory and Test-Time Compute Scaling","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2510.18471","citing_title":"CodeRL+: Improving Code Generation via Reinforcement with Execution Semantics Alignment","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2409.12917","citing_title":"Training Language Models to Self-Correct via Reinforcement Learning","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2512.07461","citing_title":"Native Parallel Reasoner: Reasoning in Parallelism via Self-Distilled Reinforcement Learning","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16382","citing_title":"LiFT: Does Instruction Fine-Tuning Improve In-Context Learning for Longitudinal Modelling by Large Language Models?","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11746","citing_title":"When Reasoning Traces Become Performative: Step-Level Evidence that Chain-of-Thought Is an Imperfect Oversight Channel","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11328","citing_title":"Epistemic Uncertainty for Test-Time Discovery","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08221","citing_title":"NoisyCoconut: Counterfactual Consensus via Latent Space Reasoning","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05893","citing_title":"Logic-Regularized Verifier Elicits Reasoning from LLMs","ref_index":72,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21268","citing_title":"Measure Twice, Click Once: Co-evolving Proposer and Visual Critic via Reinforcement Learning for GUI Grounding","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2412.06769","citing_title":"Training Large Language Models to Reason in a Continuous Latent Space","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08299","citing_title":"SeLaR: Selective Latent Reasoning in Large Language Models","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05226","citing_title":"Internalizing Outcome Supervision into Process Supervision: A New Paradigm for Reinforcement Learning for Reasoning","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RZQ36VSC34G52VNLTYNJLNG32I","json":"https://pith.science/pith/RZQ36VSC34G52VNLTYNJLNG32I.json","graph_json":"https://pith.science/api/pith-number/RZQ36VSC34G52VNLTYNJLNG32I/graph.json","events_json":"https://pith.science/api/pith-number/RZQ36VSC34G52VNLTYNJLNG32I/events.json","paper":"https://pith.science/paper/RZQ36VSC"},"agent_actions":{"view_html":"https://pith.science/pith/RZQ36VSC34G52VNLTYNJLNG32I","download_json":"https://pith.science/pith/RZQ36VSC34G52VNLTYNJLNG32I.json","view_paper":"https://pith.science/paper/RZQ36VSC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.04642&json=true","fetch_graph":"https://pith.science/api/pith-number/RZQ36VSC34G52VNLTYNJLNG32I/graph.json","fetch_events":"https://pith.science/api/pith-number/RZQ36VSC34G52VNLTYNJLNG32I/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RZQ36VSC34G52VNLTYNJLNG32I/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RZQ36VSC34G52VNLTYNJLNG32I/action/storage_attestation","attest_author":"https://pith.science/pith/RZQ36VSC34G52VNLTYNJLNG32I/action/author_attestation","sign_citation":"https://pith.science/pith/RZQ36VSC34G52VNLTYNJLNG32I/action/citation_signature","submit_replication":"https://pith.science/pith/RZQ36VSC34G52VNLTYNJLNG32I/action/replication_record"}},"created_at":"2026-07-05T07:53:26.493495+00:00","updated_at":"2026-07-05T07:53:26.493495+00:00"}