{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:R6VFQURKQYSBR2X5SHTS6JDJJK","short_pith_number":"pith:R6VFQURK","schema_version":"1.0","canonical_sha256":"8faa58522a862418eafd91e72f24694a8f1e60cc7775d94e18bae6312072fa64","source":{"kind":"arxiv","id":"2505.23585","version":2},"attestation_state":"computed","paper":{"title":"On-Policy RL with Optimal Reward Baseline","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Furu Wei, Li Dong, Shaohan Huang, Xun Wu, Yaru Hao, Zewen Chi","submitted_at":"2025-05-29T15:58:04Z","abstract_excerpt":"Reinforcement learning algorithms are fundamental to align large language models with human preferences and to enhance their reasoning capabilities. However, current reinforcement learning algorithms often suffer from training instability due to loose on-policy constraints and computational inefficiency due to auxiliary models. In this work, we propose On-Policy RL with Optimal reward baseline (OPO), a novel and simplified reinforcement learning algorithm designed to address these challenges. OPO emphasizes the importance of exact on-policy training, which empirically stabilizes the training p"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.23585","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-05-29T15:58:04Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"585c60f21382c0e814e8efe97e06de00aa8d6c9592975495bc98e5944f155ca9","abstract_canon_sha256":"5cf6ceb2f728a259142456ddfc712569944f88795ca11cb00ea6bb6b9a840817"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:15:21.547283Z","signature_b64":"6Z2DWed7JqaoIju2gtzlG4eEzzm2g35GjrJ6Oa/uYzoe/LJ+Yshia8cxSjRbNEkGT/NtC9T2yi65EwBtV1ZIAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8faa58522a862418eafd91e72f24694a8f1e60cc7775d94e18bae6312072fa64","last_reissued_at":"2026-07-05T11:15:21.546816Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:15:21.546816Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On-Policy RL with Optimal Reward Baseline","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Furu Wei, Li Dong, Shaohan Huang, Xun Wu, Yaru Hao, Zewen Chi","submitted_at":"2025-05-29T15:58:04Z","abstract_excerpt":"Reinforcement learning algorithms are fundamental to align large language models with human preferences and to enhance their reasoning capabilities. However, current reinforcement learning algorithms often suffer from training instability due to loose on-policy constraints and computational inefficiency due to auxiliary models. In this work, we propose On-Policy RL with Optimal reward baseline (OPO), a novel and simplified reinforcement learning algorithm designed to address these challenges. OPO emphasizes the importance of exact on-policy training, which empirically stabilizes the training p"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.23585","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.23585/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.23585","created_at":"2026-07-05T11:15:21.546880+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.23585v2","created_at":"2026-07-05T11:15:21.546880+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.23585","created_at":"2026-07-05T11:15:21.546880+00:00"},{"alias_kind":"pith_short_12","alias_value":"R6VFQURKQYSB","created_at":"2026-07-05T11:15:21.546880+00:00"},{"alias_kind":"pith_short_16","alias_value":"R6VFQURKQYSBR2X5","created_at":"2026-07-05T11:15:21.546880+00:00"},{"alias_kind":"pith_short_8","alias_value":"R6VFQURK","created_at":"2026-07-05T11:15:21.546880+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21943","citing_title":"Modularized Reinforcement Learning on LLMs: From MDP Creation to Exploration and Learning","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12634","citing_title":"Keep Policy Gradient in Charge: Sibling-Guided Credit Distillation for Long-Horizon Tool-Use Agents","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12634","citing_title":"Keep Policy Gradient in Charge: Sibling-Guided Credit Distillation for Long-Horizon Tool-Use Agents","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12058","citing_title":"Holder Policy Optimisation","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2604.28005","citing_title":"Kernelized Advantage Estimation: From Nonparametric Statistics to LLM Reasoning","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2510.10150","citing_title":"Rethinking Entropy Interventions in RLVR: An Entropy Change Perspective","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.00860","citing_title":"Policy Improvement Reinforcement Learning","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12058","citing_title":"Holder Policy Optimisation","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11491","citing_title":"Understanding and Preventing Entropy Collapse in RLVR with On-Policy Entropy Flow Optimization","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2604.28005","citing_title":"Kernelized Advantage Estimation: From Nonparametric Statistics to LLM Reasoning","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05965","citing_title":"Beyond Uniform Credit Assignment: Selective Eligibility Traces for RLVR","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/R6VFQURKQYSBR2X5SHTS6JDJJK","json":"https://pith.science/pith/R6VFQURKQYSBR2X5SHTS6JDJJK.json","graph_json":"https://pith.science/api/pith-number/R6VFQURKQYSBR2X5SHTS6JDJJK/graph.json","events_json":"https://pith.science/api/pith-number/R6VFQURKQYSBR2X5SHTS6JDJJK/events.json","paper":"https://pith.science/paper/R6VFQURK"},"agent_actions":{"view_html":"https://pith.science/pith/R6VFQURKQYSBR2X5SHTS6JDJJK","download_json":"https://pith.science/pith/R6VFQURKQYSBR2X5SHTS6JDJJK.json","view_paper":"https://pith.science/paper/R6VFQURK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.23585&json=true","fetch_graph":"https://pith.science/api/pith-number/R6VFQURKQYSBR2X5SHTS6JDJJK/graph.json","fetch_events":"https://pith.science/api/pith-number/R6VFQURKQYSBR2X5SHTS6JDJJK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/R6VFQURKQYSBR2X5SHTS6JDJJK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/R6VFQURKQYSBR2X5SHTS6JDJJK/action/storage_attestation","attest_author":"https://pith.science/pith/R6VFQURKQYSBR2X5SHTS6JDJJK/action/author_attestation","sign_citation":"https://pith.science/pith/R6VFQURKQYSBR2X5SHTS6JDJJK/action/citation_signature","submit_replication":"https://pith.science/pith/R6VFQURKQYSBR2X5SHTS6JDJJK/action/replication_record"}},"created_at":"2026-07-05T11:15:21.546880+00:00","updated_at":"2026-07-05T11:15:21.546880+00:00"}