{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:QFOVJHUN4UDBMGQYGMHDPWWHZA","short_pith_number":"pith:QFOVJHUN","schema_version":"1.0","canonical_sha256":"815d549e8de506161a18330e37dac7c800370fa2c87d1bcc9f7c99db0a6cb9c7","source":{"kind":"arxiv","id":"2506.13923","version":2},"attestation_state":"computed","paper":{"title":"Adaptive Guidance Accelerates Reinforcement Learning of Reasoning Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Anisha Gunjal, Elaine Lau, Manasi Sharma, Nikhil Baharte, Sean Hendryx, Vaskar Nath","submitted_at":"2025-06-16T19:03:06Z","abstract_excerpt":"We study the process through which reasoning models trained with reinforcement learning on verifiable rewards (RLVR) can learn to solve new problems. We find that RLVR drives performance in two main ways: (1) by compressing pass@$k$ into pass@1 and (2) via \"capability gain\" in which models learn to solve new problems that they previously could not solve even at high $k$. We find that while capability gain exists across model scales, learning to solve new problems is primarily driven through self-distillation. We demonstrate these findings across model scales ranging from 0.5B to 72B parameters"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.13923","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-16T19:03:06Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"cbd84d83c90de728f71e0490e56e738f645bb0471e7596d224f2f43443636bf1","abstract_canon_sha256":"4678b2d50b7e87470347bebdbd892549fa38574b4e39ec71f26a714e03ef0c56"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:24:23.973773Z","signature_b64":"n1mFZWVweYQw5gSNyW2NokTNYUQYOHSzsdarf8dA6Tgj1SswiGtt7XJnHXqPsZ9V4p8ZDqUzebDPeDRnmcrmBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"815d549e8de506161a18330e37dac7c800370fa2c87d1bcc9f7c99db0a6cb9c7","last_reissued_at":"2026-07-05T11:24:23.973167Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:24:23.973167Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Adaptive Guidance Accelerates Reinforcement Learning of Reasoning Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Anisha Gunjal, Elaine Lau, Manasi Sharma, Nikhil Baharte, Sean Hendryx, Vaskar Nath","submitted_at":"2025-06-16T19:03:06Z","abstract_excerpt":"We study the process through which reasoning models trained with reinforcement learning on verifiable rewards (RLVR) can learn to solve new problems. We find that RLVR drives performance in two main ways: (1) by compressing pass@$k$ into pass@1 and (2) via \"capability gain\" in which models learn to solve new problems that they previously could not solve even at high $k$. We find that while capability gain exists across model scales, learning to solve new problems is primarily driven through self-distillation. We demonstrate these findings across model scales ranging from 0.5B to 72B parameters"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.13923","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.13923/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.13923","created_at":"2026-07-05T11:24:23.973244+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.13923v2","created_at":"2026-07-05T11:24:23.973244+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.13923","created_at":"2026-07-05T11:24:23.973244+00:00"},{"alias_kind":"pith_short_12","alias_value":"QFOVJHUN4UDB","created_at":"2026-07-05T11:24:23.973244+00:00"},{"alias_kind":"pith_short_16","alias_value":"QFOVJHUN4UDBMGQY","created_at":"2026-07-05T11:24:23.973244+00:00"},{"alias_kind":"pith_short_8","alias_value":"QFOVJHUN","created_at":"2026-07-05T11:24:23.973244+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18216","citing_title":"Zone of Proximal Policy Optimization: Teacher in Prompts, Not Gradients","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25198","citing_title":"Hide to Guide: Learning via Semantic Masking","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2508.07809","citing_title":"EvoCoT: Overcoming the Exploration Bottleneck in Reinforcement Learning","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2509.21882","citing_title":"Position: The Hidden Costs and Measurement Gaps of Reinforcement Learning with Verifiable Rewards","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11505","citing_title":"Selective Off-Policy Reference Tuning with Plan Guidance","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12004","citing_title":"Learning Agentic Policy from Action Guidance","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11505","citing_title":"Selective Off-Policy Reference Tuning with Plan Guidance","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QFOVJHUN4UDBMGQYGMHDPWWHZA","json":"https://pith.science/pith/QFOVJHUN4UDBMGQYGMHDPWWHZA.json","graph_json":"https://pith.science/api/pith-number/QFOVJHUN4UDBMGQYGMHDPWWHZA/graph.json","events_json":"https://pith.science/api/pith-number/QFOVJHUN4UDBMGQYGMHDPWWHZA/events.json","paper":"https://pith.science/paper/QFOVJHUN"},"agent_actions":{"view_html":"https://pith.science/pith/QFOVJHUN4UDBMGQYGMHDPWWHZA","download_json":"https://pith.science/pith/QFOVJHUN4UDBMGQYGMHDPWWHZA.json","view_paper":"https://pith.science/paper/QFOVJHUN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.13923&json=true","fetch_graph":"https://pith.science/api/pith-number/QFOVJHUN4UDBMGQYGMHDPWWHZA/graph.json","fetch_events":"https://pith.science/api/pith-number/QFOVJHUN4UDBMGQYGMHDPWWHZA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QFOVJHUN4UDBMGQYGMHDPWWHZA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QFOVJHUN4UDBMGQYGMHDPWWHZA/action/storage_attestation","attest_author":"https://pith.science/pith/QFOVJHUN4UDBMGQYGMHDPWWHZA/action/author_attestation","sign_citation":"https://pith.science/pith/QFOVJHUN4UDBMGQYGMHDPWWHZA/action/citation_signature","submit_replication":"https://pith.science/pith/QFOVJHUN4UDBMGQYGMHDPWWHZA/action/replication_record"}},"created_at":"2026-07-05T11:24:23.973244+00:00","updated_at":"2026-07-05T11:24:23.973244+00:00"}