{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:BIPLZJ4YYKRK7FUPTDLJRTEYNS","short_pith_number":"pith:BIPLZJ4Y","schema_version":"1.0","canonical_sha256":"0a1ebca798c2a2af968f98d698cc986ca4aefa38fb86423124ac16bf1df26d84","source":{"kind":"arxiv","id":"2402.05808","version":2},"attestation_state":"computed","paper":{"title":"Training Large Language Models for Reasoning through Reverse Curriculum Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.AI","authors_text":"Boyang Hong, Honglin Guo, Junzhe Wang, Peng Sun, Qi Zhang, Rui Zheng, Senjie Jin, Shichun Liu, Shihan Dou, Tao Gui, Wei He, Wei Shen, Wenxiang Chen, Xiaoran Fan, Xiao Wang, Xinbo Zhang, Xin Guo, Xuanjing Huang, Yiwen Ding, Yuhao Zhou, Zhiheng Xi","submitted_at":"2024-02-08T16:46:26Z","abstract_excerpt":"In this paper, we propose R$^3$: Learning Reasoning through Reverse Curriculum Reinforcement Learning (RL), a novel method that employs only outcome supervision to achieve the benefits of process supervision for large language models. The core challenge in applying RL to complex reasoning is to identify a sequence of actions that result in positive rewards and provide appropriate supervision for optimization. Outcome supervision provides sparse rewards for final results without identifying error locations, whereas process supervision offers step-wise rewards but requires extensive manual annot"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.05808","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2024-02-08T16:46:26Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"c79902916a6340d141f4d03f4f4448ee7d3fac126f16a325864c85707f89cd5b","abstract_canon_sha256":"166f79e518e833fdbfa2c277e40d5b60157a175316e9703f23e354d16e4c5faf"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:56:59.393678Z","signature_b64":"cVzkRWmsFbqmM5JE67Qx1h7QSOfLKD7tMg7rviwPDsME5WRGdbuYNqzTUuioCe+e60MnETn6MjuCH358CICsDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0a1ebca798c2a2af968f98d698cc986ca4aefa38fb86423124ac16bf1df26d84","last_reissued_at":"2026-07-05T07:56:59.393173Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:56:59.393173Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Training Large Language Models for Reasoning through Reverse Curriculum Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.AI","authors_text":"Boyang Hong, Honglin Guo, Junzhe Wang, Peng Sun, Qi Zhang, Rui Zheng, Senjie Jin, Shichun Liu, Shihan Dou, Tao Gui, Wei He, Wei Shen, Wenxiang Chen, Xiaoran Fan, Xiao Wang, Xinbo Zhang, Xin Guo, Xuanjing Huang, Yiwen Ding, Yuhao Zhou, Zhiheng Xi","submitted_at":"2024-02-08T16:46:26Z","abstract_excerpt":"In this paper, we propose R$^3$: Learning Reasoning through Reverse Curriculum Reinforcement Learning (RL), a novel method that employs only outcome supervision to achieve the benefits of process supervision for large language models. The core challenge in applying RL to complex reasoning is to identify a sequence of actions that result in positive rewards and provide appropriate supervision for optimization. Outcome supervision provides sparse rewards for final results without identifying error locations, whereas process supervision offers step-wise rewards but requires extensive manual annot"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.05808","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.05808/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.05808","created_at":"2026-07-05T07:56:59.393241+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.05808v2","created_at":"2026-07-05T07:56:59.393241+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.05808","created_at":"2026-07-05T07:56:59.393241+00:00"},{"alias_kind":"pith_short_12","alias_value":"BIPLZJ4YYKRK","created_at":"2026-07-05T07:56:59.393241+00:00"},{"alias_kind":"pith_short_16","alias_value":"BIPLZJ4YYKRK7FUP","created_at":"2026-07-05T07:56:59.393241+00:00"},{"alias_kind":"pith_short_8","alias_value":"BIPLZJ4Y","created_at":"2026-07-05T07:56:59.393241+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07674","citing_title":"Max Out GRPO Signal: Adaptive Trace Prefix Control for Hard Reasoning Problems","ref_index":8,"is_internal_anchor":true},{"citing_arxiv_id":"2606.21943","citing_title":"Modularized Reinforcement Learning on LLMs: From MDP Creation to Exploration and Learning","ref_index":229,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00564","citing_title":"Decomposed On-Policy Distillation for Vision-Language Reasoning: Steering Gradients for Visual Grounding","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2411.10442","citing_title":"Enhancing the Reasoning Ability of Multimodal Large Language Models via Mixed Preference Optimization","ref_index":106,"is_internal_anchor":false},{"citing_arxiv_id":"2602.14868","citing_title":"Goldilocks RL: Tuning Task Difficulty to Escape Sparse Rewards for Reasoning","ref_index":30,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BIPLZJ4YYKRK7FUPTDLJRTEYNS","json":"https://pith.science/pith/BIPLZJ4YYKRK7FUPTDLJRTEYNS.json","graph_json":"https://pith.science/api/pith-number/BIPLZJ4YYKRK7FUPTDLJRTEYNS/graph.json","events_json":"https://pith.science/api/pith-number/BIPLZJ4YYKRK7FUPTDLJRTEYNS/events.json","paper":"https://pith.science/paper/BIPLZJ4Y"},"agent_actions":{"view_html":"https://pith.science/pith/BIPLZJ4YYKRK7FUPTDLJRTEYNS","download_json":"https://pith.science/pith/BIPLZJ4YYKRK7FUPTDLJRTEYNS.json","view_paper":"https://pith.science/paper/BIPLZJ4Y","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.05808&json=true","fetch_graph":"https://pith.science/api/pith-number/BIPLZJ4YYKRK7FUPTDLJRTEYNS/graph.json","fetch_events":"https://pith.science/api/pith-number/BIPLZJ4YYKRK7FUPTDLJRTEYNS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BIPLZJ4YYKRK7FUPTDLJRTEYNS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BIPLZJ4YYKRK7FUPTDLJRTEYNS/action/storage_attestation","attest_author":"https://pith.science/pith/BIPLZJ4YYKRK7FUPTDLJRTEYNS/action/author_attestation","sign_citation":"https://pith.science/pith/BIPLZJ4YYKRK7FUPTDLJRTEYNS/action/citation_signature","submit_replication":"https://pith.science/pith/BIPLZJ4YYKRK7FUPTDLJRTEYNS/action/replication_record"}},"created_at":"2026-07-05T07:56:59.393241+00:00","updated_at":"2026-07-05T07:56:59.393241+00:00"}