{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:YWBKCDXFDLOMJUSKPO4ZSOMGDD","short_pith_number":"pith:YWBKCDXF","schema_version":"1.0","canonical_sha256":"c582a10ee51adcc4d24a7bb999398618c5bde3ba0d4538119158408657946d5e","source":{"kind":"arxiv","id":"2608.03068","version":1},"attestation_state":"computed","paper":{"title":"CVPO: Enhancing LLM Reinforcement Learning Reasoning via Value-Variance Adaptation and Dynamic Curriculum Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Bo Pang, Hangfei Xu, Panpan Li, Shengzhao Wen, Shiyong Li, Yalu Ouyang, Yanpeng Wang, Ziqi Jia","submitted_at":"2026-08-04T03:30:59Z","abstract_excerpt":"Reinforcement learning (RL) has emerged as an effective method for enhancing the reasoning capabilities of large language models (LLMs). However, existing methods suffer from insufficient precision in feedback on generated answer trajectories and exhibit the phenomenon of problem difficulty drift. To address these challenges, we propose CVPO - Curriculum-guided Value-Variance Policy Optimization. At the response trajectory level, we find that token-level value-variance correlates with exploration intensity. Our theoretical analysis shows this variance bounds policy update magnitude. We then us"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2608.03068","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2026-08-04T03:30:59Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"22c6711b9691e768c9cc82dda07cd6e08032444ec7678ece5c12f4f0525fc052","abstract_canon_sha256":"656d2f5d3beab42ce84a39b6c10ad31634d33b10075b2704af5e62ab9fa6c753"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-05T00:45:05.880059Z","signature_b64":"74ztXLkSMDiVMdhDJQQfkl5NAY5Y13DD851w+A5lRefmt9yPwPM8P07IzRV2J1ueb6Mem7eqyq8oa7tdxu3vAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c582a10ee51adcc4d24a7bb999398618c5bde3ba0d4538119158408657946d5e","last_reissued_at":"2026-08-05T00:45:05.877639Z","signature_status":"signed_v1","first_computed_at":"2026-08-05T00:45:05.877639Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CVPO: Enhancing LLM Reinforcement Learning Reasoning via Value-Variance Adaptation and Dynamic Curriculum Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Bo Pang, Hangfei Xu, Panpan Li, Shengzhao Wen, Shiyong Li, Yalu Ouyang, Yanpeng Wang, Ziqi Jia","submitted_at":"2026-08-04T03:30:59Z","abstract_excerpt":"Reinforcement learning (RL) has emerged as an effective method for enhancing the reasoning capabilities of large language models (LLMs). However, existing methods suffer from insufficient precision in feedback on generated answer trajectories and exhibit the phenomenon of problem difficulty drift. To address these challenges, we propose CVPO - Curriculum-guided Value-Variance Policy Optimization. At the response trajectory level, we find that token-level value-variance correlates with exploration intensity. Our theoretical analysis shows this variance bounds policy update magnitude. We then us"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2608.03068","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2608.03068/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2608.03068","created_at":"2026-08-05T00:45:05.878532+00:00"},{"alias_kind":"arxiv_version","alias_value":"2608.03068v1","created_at":"2026-08-05T00:45:05.878532+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2608.03068","created_at":"2026-08-05T00:45:05.878532+00:00"},{"alias_kind":"pith_short_12","alias_value":"YWBKCDXFDLOM","created_at":"2026-08-05T00:45:05.878532+00:00"},{"alias_kind":"pith_short_16","alias_value":"YWBKCDXFDLOMJUSK","created_at":"2026-08-05T00:45:05.878532+00:00"},{"alias_kind":"pith_short_8","alias_value":"YWBKCDXF","created_at":"2026-08-05T00:45:05.878532+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YWBKCDXFDLOMJUSKPO4ZSOMGDD","json":"https://pith.science/pith/YWBKCDXFDLOMJUSKPO4ZSOMGDD.json","graph_json":"https://pith.science/api/pith-number/YWBKCDXFDLOMJUSKPO4ZSOMGDD/graph.json","events_json":"https://pith.science/api/pith-number/YWBKCDXFDLOMJUSKPO4ZSOMGDD/events.json","paper":"https://pith.science/paper/YWBKCDXF"},"agent_actions":{"view_html":"https://pith.science/pith/YWBKCDXFDLOMJUSKPO4ZSOMGDD","download_json":"https://pith.science/pith/YWBKCDXFDLOMJUSKPO4ZSOMGDD.json","view_paper":"https://pith.science/paper/YWBKCDXF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2608.03068&json=true","fetch_graph":"https://pith.science/api/pith-number/YWBKCDXFDLOMJUSKPO4ZSOMGDD/graph.json","fetch_events":"https://pith.science/api/pith-number/YWBKCDXFDLOMJUSKPO4ZSOMGDD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YWBKCDXFDLOMJUSKPO4ZSOMGDD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YWBKCDXFDLOMJUSKPO4ZSOMGDD/action/storage_attestation","attest_author":"https://pith.science/pith/YWBKCDXFDLOMJUSKPO4ZSOMGDD/action/author_attestation","sign_citation":"https://pith.science/pith/YWBKCDXFDLOMJUSKPO4ZSOMGDD/action/citation_signature","submit_replication":"https://pith.science/pith/YWBKCDXFDLOMJUSKPO4ZSOMGDD/action/replication_record"}},"created_at":"2026-08-05T00:45:05.878532+00:00","updated_at":"2026-08-05T00:45:05.878532+00:00"}