{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:FKUZSAERHR7F2S76XZQHE5I2Q6","short_pith_number":"pith:FKUZSAER","schema_version":"1.0","canonical_sha256":"2aa99900913c7e5d4bfebe6072751a87af1648597b9f6eb02de785da1e6e52ca","source":{"kind":"arxiv","id":"2401.08967","version":3},"attestation_state":"computed","paper":{"title":"ReFT: Reasoning with Reinforced Fine-Tuning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hang Li, Peng Sun, Trung Quoc Luong, Xiaoran Jin, Xinbo Zhang, Zhanming Jie","submitted_at":"2024-01-17T04:43:21Z","abstract_excerpt":"One way to enhance the reasoning capability of Large Language Models (LLMs) is to conduct Supervised Fine-Tuning (SFT) using Chain-of-Thought (CoT) annotations. This approach does not show sufficiently strong generalization ability, however, because the training only relies on the given CoT data. In math problem-solving, for example, there is usually only one annotated reasoning path for each question in the training data. Intuitively, it would be better for the algorithm to learn from multiple annotated reasoning paths given a question. To address this issue, we propose a simple yet effective"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.08967","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-01-17T04:43:21Z","cross_cats_sorted":[],"title_canon_sha256":"7079054bb8e87f6a13bea6b22945ce55be500dee155c5d42727f3279a6c43525","abstract_canon_sha256":"2c0607116b51589f14ca0982ff9803b6f56f3cdaeeebb0f9620371d82977bc82"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:48:30.904833Z","signature_b64":"NPbgBnb74JKpD+pkq9kk0G7u5qGEYJfsmXL8F/vQVNHUjnBucDx+IBtUZ/KDthYm6TNvN7yLvQXzO/oXGwNJBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2aa99900913c7e5d4bfebe6072751a87af1648597b9f6eb02de785da1e6e52ca","last_reissued_at":"2026-07-05T09:48:30.904322Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:48:30.904322Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ReFT: Reasoning with Reinforced Fine-Tuning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hang Li, Peng Sun, Trung Quoc Luong, Xiaoran Jin, Xinbo Zhang, Zhanming Jie","submitted_at":"2024-01-17T04:43:21Z","abstract_excerpt":"One way to enhance the reasoning capability of Large Language Models (LLMs) is to conduct Supervised Fine-Tuning (SFT) using Chain-of-Thought (CoT) annotations. This approach does not show sufficiently strong generalization ability, however, because the training only relies on the given CoT data. In math problem-solving, for example, there is usually only one annotated reasoning path for each question in the training data. Intuitively, it would be better for the algorithm to learn from multiple annotated reasoning paths given a question. To address this issue, we propose a simple yet effective"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.08967","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.08967/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.08967","created_at":"2026-07-05T09:48:30.904386+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.08967v3","created_at":"2026-07-05T09:48:30.904386+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.08967","created_at":"2026-07-05T09:48:30.904386+00:00"},{"alias_kind":"pith_short_12","alias_value":"FKUZSAERHR7F","created_at":"2026-07-05T09:48:30.904386+00:00"},{"alias_kind":"pith_short_16","alias_value":"FKUZSAERHR7F2S76","created_at":"2026-07-05T09:48:30.904386+00:00"},{"alias_kind":"pith_short_8","alias_value":"FKUZSAER","created_at":"2026-07-05T09:48:30.904386+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":21,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22158","citing_title":"Improving Reasoning in Vision-Language Models via Perception Verified Self-Training","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21228","citing_title":"Sakana Fugu Technical Report","ref_index":290,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22158","citing_title":"Improving Reasoning in Vision-Language Models via Perception Verified Self-Training","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25198","citing_title":"Hide to Guide: Learning via Semantic Masking","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2504.02181","citing_title":"A Survey of Scaling in Large Language Model Reasoning","ref_index":128,"is_internal_anchor":false},{"citing_arxiv_id":"2602.07832","citing_title":"rePIRL: Learn PRM with Inverse RL for LLM Reasoning","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06638","citing_title":"Can RL Teach Long-Horizon Reasoning to LLMs? Expressiveness Is Key","ref_index":87,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17187","citing_title":"PluRule: A Benchmark for Moderating Pluralistic Communities on Social Media","ref_index":88,"is_internal_anchor":false},{"citing_arxiv_id":"2505.15134","citing_title":"The Unreasonable Effectiveness of Entropy Minimization in LLM Reasoning","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2509.21882","citing_title":"Position: The Hidden Costs and Measurement Gaps of Reinforcement Learning with Verifiable Rewards","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2511.11113","citing_title":"VIDEOP2R: Video Understanding from Perception to Reasoning","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2503.12937","citing_title":"R1-VL: Learning to Reason with Multimodal Large Language Models via Step-wise Group Relative Policy Optimization","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2601.13262","citing_title":"CURE-Med: Curriculum-Informed Reinforcement Learning for Multilingual Medical Reasoning","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2412.18925","citing_title":"HuatuoGPT-o1, Towards Medical Complex Reasoning with LLMs","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2504.11536","citing_title":"ReTool: Reinforcement Learning for Strategic Tool Use in LLMs","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08516","citing_title":"OracleTSC: Oracle-Informed Reward Hurdle and Uncertainty Regularization for Traffic Signal Control","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09262","citing_title":"Reinforcing Multimodal Reasoning Against Visual Degradation","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06638","citing_title":"Can RL Teach Long-Horizon Reasoning to LLMs? Expressiveness Is Key","ref_index":87,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06638","citing_title":"Can RL Teach Long-Horizon Reasoning to LLMs? Expressiveness Is Key","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21138","citing_title":"Navigating the Clutter: Waypoint-Based Bi-Level Planning for Multi-Robot Systems","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18327","citing_title":"PARM: Pipeline-Adapted Reward Model","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FKUZSAERHR7F2S76XZQHE5I2Q6","json":"https://pith.science/pith/FKUZSAERHR7F2S76XZQHE5I2Q6.json","graph_json":"https://pith.science/api/pith-number/FKUZSAERHR7F2S76XZQHE5I2Q6/graph.json","events_json":"https://pith.science/api/pith-number/FKUZSAERHR7F2S76XZQHE5I2Q6/events.json","paper":"https://pith.science/paper/FKUZSAER"},"agent_actions":{"view_html":"https://pith.science/pith/FKUZSAERHR7F2S76XZQHE5I2Q6","download_json":"https://pith.science/pith/FKUZSAERHR7F2S76XZQHE5I2Q6.json","view_paper":"https://pith.science/paper/FKUZSAER","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.08967&json=true","fetch_graph":"https://pith.science/api/pith-number/FKUZSAERHR7F2S76XZQHE5I2Q6/graph.json","fetch_events":"https://pith.science/api/pith-number/FKUZSAERHR7F2S76XZQHE5I2Q6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FKUZSAERHR7F2S76XZQHE5I2Q6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FKUZSAERHR7F2S76XZQHE5I2Q6/action/storage_attestation","attest_author":"https://pith.science/pith/FKUZSAERHR7F2S76XZQHE5I2Q6/action/author_attestation","sign_citation":"https://pith.science/pith/FKUZSAERHR7F2S76XZQHE5I2Q6/action/citation_signature","submit_replication":"https://pith.science/pith/FKUZSAERHR7F2S76XZQHE5I2Q6/action/replication_record"}},"created_at":"2026-07-05T09:48:30.904386+00:00","updated_at":"2026-07-05T09:48:30.904386+00:00"}