{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:N4CGYDDUJ5I2Y4ISHAHVB7KZST","short_pith_number":"pith:N4CGYDDU","schema_version":"1.0","canonical_sha256":"6f046c0c744f51ac7112380f50fd5994eb24e195bbc1b96c2ce74cc1daf03369","source":{"kind":"arxiv","id":"2503.01491","version":1},"attestation_state":"computed","paper":{"title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Lin Yan, Ruofei Zhu, Tiantian Fan, Yufeng Yuan, Yu Yue","submitted_at":"2025-03-03T12:59:25Z","abstract_excerpt":"Reinforcement learning (RL) is pivotal for enabling large language models (LLMs) to generate long chains of thought (CoT) for complex tasks like math and reasoning. However, Proximal Policy Optimization (PPO), effective in many RL scenarios, fails in long CoT tasks. This paper identifies that value initialization bias and reward signal decay are the root causes of PPO's failure. We propose Value-Calibrated PPO (VC-PPO) to address these issues. In VC-PPO, the value model is pretrained to tackle initialization bias, and the Generalized Advantage Estimation (GAE) computation is decoupled between "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.01491","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-03-03T12:59:25Z","cross_cats_sorted":[],"title_canon_sha256":"917f49a5a0f0fabd246234e660e917a9f936d7d0a5fbe1bddb6d95892d1cc83a","abstract_canon_sha256":"6bcdfab7ba3f13d8f4f9784d819a0dcda2d5dbbea9a92ca4acc4653badf59766"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:23:08.333696Z","signature_b64":"iXeHKXZ0UFgXkC3LOFVqM+YPZCNHXnkFe9k0yRVP29fo4jVgBQHxKaPoI/mLpdlkg34ZwtBmCczNcTiAkjW4Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6f046c0c744f51ac7112380f50fd5994eb24e195bbc1b96c2ce74cc1daf03369","last_reissued_at":"2026-07-05T10:23:08.332938Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:23:08.332938Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"What's Behind PPO's Collapse in Long-CoT? Value Optimization Holds the Secret","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Lin Yan, Ruofei Zhu, Tiantian Fan, Yufeng Yuan, Yu Yue","submitted_at":"2025-03-03T12:59:25Z","abstract_excerpt":"Reinforcement learning (RL) is pivotal for enabling large language models (LLMs) to generate long chains of thought (CoT) for complex tasks like math and reasoning. However, Proximal Policy Optimization (PPO), effective in many RL scenarios, fails in long CoT tasks. This paper identifies that value initialization bias and reward signal decay are the root causes of PPO's failure. We propose Value-Calibrated PPO (VC-PPO) to address these issues. In VC-PPO, the value model is pretrained to tackle initialization bias, and the Generalized Advantage Estimation (GAE) computation is decoupled between "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.01491","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.01491/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.01491","created_at":"2026-07-05T10:23:08.333034+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.01491v1","created_at":"2026-07-05T10:23:08.333034+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.01491","created_at":"2026-07-05T10:23:08.333034+00:00"},{"alias_kind":"pith_short_12","alias_value":"N4CGYDDUJ5I2","created_at":"2026-07-05T10:23:08.333034+00:00"},{"alias_kind":"pith_short_16","alias_value":"N4CGYDDUJ5I2Y4IS","created_at":"2026-07-05T10:23:08.333034+00:00"},{"alias_kind":"pith_short_8","alias_value":"N4CGYDDU","created_at":"2026-07-05T10:23:08.333034+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":28,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2607.05378","citing_title":"CompactionRL: Reinforcement Learning with Context Compaction for Long-Horizon Agents","ref_index":21,"is_internal_anchor":true},{"citing_arxiv_id":"2607.05378","citing_title":"CompactionRL: Reinforcement Learning with Context Compaction for Long-Horizon Agents","ref_index":22,"is_internal_anchor":true},{"citing_arxiv_id":"2606.21943","citing_title":"Modularized Reinforcement Learning on LLMs: From MDP Creation to Exploration and Learning","ref_index":254,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20008","citing_title":"VIMPO: Value-Implicit Policy Optimization for LLMs","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03102","citing_title":"Small RL Controller, Large Language Model: RL-Guided Adaptive Sampling for Test-Time Scaling","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01281","citing_title":"RLVR without Ineffective Samples: Group Prioritized Off-Policy Optimization for LLM Reasoning","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01249","citing_title":"Trust Region On-Policy Distillation","ref_index":207,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25582","citing_title":"Extreme Region Policy Distillation","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2503.14476","citing_title":"DAPO: An Open-Source LLM Reinforcement Learning System at Scale","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2504.12501","citing_title":"Reinforcement Learning from Human Feedback","ref_index":149,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12058","citing_title":"Holder Policy Optimisation","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2507.01679","citing_title":"Blending Supervised and Reinforcement Fine-Tuning with Prefix Sampling","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2602.01970","citing_title":"Small Generalizable Prompt Predictive Models Can Steer Efficient RL Post-Training of Large Reasoning Models","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2510.00568","citing_title":"ReSeek: A Self-Correcting Framework for Search Agents with Instructive Rewards","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2511.07833","citing_title":"MURPHY: Feedback-Aware GRPO with Retrospective Credit Assignment for Multi-Turn Code Generation","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2510.13786","citing_title":"The Art of Scaling Reinforcement Learning Compute for LLMs","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2504.20571","citing_title":"Reinforcement Learning for Reasoning in Large Language Models with One Training Example","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2509.02544","citing_title":"UI-TARS-2 Technical Report: Advancing GUI Agent with Multi-Turn Reinforcement Learning","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2504.05118","citing_title":"VAPO: Efficient and Reliable Reinforcement Learning for Advanced Reasoning Tasks","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12058","citing_title":"Holder Policy Optimisation","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08905","citing_title":"Forge: Quality-Aware Reinforcement Learning for NP-Hard Optimization in LLMs","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01327","citing_title":"Segment-Aligned Policy Optimization for Multi-Modal Reasoning","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10701","citing_title":"Bringing Value Models Back: Generative Critics for Value Modeling in LLM Reinforcement Learning","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13010","citing_title":"Lightning OPD: Efficient Post-Training for Large Reasoning Models with Offline On-Policy Distillation","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13197","citing_title":"Unleashing Implicit Rewards: Prefix-Value Learning for Distribution-Level Optimization","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/N4CGYDDUJ5I2Y4ISHAHVB7KZST","json":"https://pith.science/pith/N4CGYDDUJ5I2Y4ISHAHVB7KZST.json","graph_json":"https://pith.science/api/pith-number/N4CGYDDUJ5I2Y4ISHAHVB7KZST/graph.json","events_json":"https://pith.science/api/pith-number/N4CGYDDUJ5I2Y4ISHAHVB7KZST/events.json","paper":"https://pith.science/paper/N4CGYDDU"},"agent_actions":{"view_html":"https://pith.science/pith/N4CGYDDUJ5I2Y4ISHAHVB7KZST","download_json":"https://pith.science/pith/N4CGYDDUJ5I2Y4ISHAHVB7KZST.json","view_paper":"https://pith.science/paper/N4CGYDDU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.01491&json=true","fetch_graph":"https://pith.science/api/pith-number/N4CGYDDUJ5I2Y4ISHAHVB7KZST/graph.json","fetch_events":"https://pith.science/api/pith-number/N4CGYDDUJ5I2Y4ISHAHVB7KZST/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/N4CGYDDUJ5I2Y4ISHAHVB7KZST/action/timestamp_anchor","attest_storage":"https://pith.science/pith/N4CGYDDUJ5I2Y4ISHAHVB7KZST/action/storage_attestation","attest_author":"https://pith.science/pith/N4CGYDDUJ5I2Y4ISHAHVB7KZST/action/author_attestation","sign_citation":"https://pith.science/pith/N4CGYDDUJ5I2Y4ISHAHVB7KZST/action/citation_signature","submit_replication":"https://pith.science/pith/N4CGYDDUJ5I2Y4ISHAHVB7KZST/action/replication_record"}},"created_at":"2026-07-05T10:23:08.333034+00:00","updated_at":"2026-07-05T10:23:08.333034+00:00"}