{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:QFOVJHUN4UDBMGQYGMHDPWWHZA","short_pith_number":"pith:QFOVJHUN","canonical_record":{"source":{"id":"2506.13923","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-16T19:03:06Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"cbd84d83c90de728f71e0490e56e738f645bb0471e7596d224f2f43443636bf1","abstract_canon_sha256":"4678b2d50b7e87470347bebdbd892549fa38574b4e39ec71f26a714e03ef0c56"},"schema_version":"1.0"},"canonical_sha256":"815d549e8de506161a18330e37dac7c800370fa2c87d1bcc9f7c99db0a6cb9c7","source":{"kind":"arxiv","id":"2506.13923","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2506.13923","created_at":"2026-07-05T11:24:23Z"},{"alias_kind":"arxiv_version","alias_value":"2506.13923v2","created_at":"2026-07-05T11:24:23Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.13923","created_at":"2026-07-05T11:24:23Z"},{"alias_kind":"pith_short_12","alias_value":"QFOVJHUN4UDB","created_at":"2026-07-05T11:24:23Z"},{"alias_kind":"pith_short_16","alias_value":"QFOVJHUN4UDBMGQY","created_at":"2026-07-05T11:24:23Z"},{"alias_kind":"pith_short_8","alias_value":"QFOVJHUN","created_at":"2026-07-05T11:24:23Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:QFOVJHUN4UDBMGQYGMHDPWWHZA","target":"record","payload":{"canonical_record":{"source":{"id":"2506.13923","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-16T19:03:06Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"cbd84d83c90de728f71e0490e56e738f645bb0471e7596d224f2f43443636bf1","abstract_canon_sha256":"4678b2d50b7e87470347bebdbd892549fa38574b4e39ec71f26a714e03ef0c56"},"schema_version":"1.0"},"canonical_sha256":"815d549e8de506161a18330e37dac7c800370fa2c87d1bcc9f7c99db0a6cb9c7","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:24:23.973773Z","signature_b64":"n1mFZWVweYQw5gSNyW2NokTNYUQYOHSzsdarf8dA6Tgj1SswiGtt7XJnHXqPsZ9V4p8ZDqUzebDPeDRnmcrmBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"815d549e8de506161a18330e37dac7c800370fa2c87d1bcc9f7c99db0a6cb9c7","last_reissued_at":"2026-07-05T11:24:23.973167Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:24:23.973167Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2506.13923","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:24:23Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"uV7qLYY36+5m5i56IRasc5BhzLQHRMDyJOwT51Nx38OAoDFdpntkQognYYPtgB8XJL+J/U/D/xwkoU6T+XuzDw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T07:17:16.080913Z"},"content_sha256":"ded41728221b80bccaeb66ebf5fdd02562a175b20177a3410b2eddb65ae4e283","schema_version":"1.0","event_id":"sha256:ded41728221b80bccaeb66ebf5fdd02562a175b20177a3410b2eddb65ae4e283"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:QFOVJHUN4UDBMGQYGMHDPWWHZA","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Adaptive Guidance Accelerates Reinforcement Learning of Reasoning Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Anisha Gunjal, Elaine Lau, Manasi Sharma, Nikhil Baharte, Sean Hendryx, Vaskar Nath","submitted_at":"2025-06-16T19:03:06Z","abstract_excerpt":"We study the process through which reasoning models trained with reinforcement learning on verifiable rewards (RLVR) can learn to solve new problems. We find that RLVR drives performance in two main ways: (1) by compressing pass@$k$ into pass@1 and (2) via \"capability gain\" in which models learn to solve new problems that they previously could not solve even at high $k$. We find that while capability gain exists across model scales, learning to solve new problems is primarily driven through self-distillation. We demonstrate these findings across model scales ranging from 0.5B to 72B parameters"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.13923","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.13923/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:24:23Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"iy1uFw/3ml8cKyoiCqF0kyxXUdKY7ExZG0B71K6OAL/BPv4CuoVFyTBPeNiBB7iwxomqCm3xnBWC5Ekrn3Y1AQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T07:17:16.081407Z"},"content_sha256":"a7949d7cd4580a990de0a6cac0c35bf6531ee008999cc41a21deb7bec06235b3","schema_version":"1.0","event_id":"sha256:a7949d7cd4580a990de0a6cac0c35bf6531ee008999cc41a21deb7bec06235b3"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/QFOVJHUN4UDBMGQYGMHDPWWHZA/bundle.json","state_url":"https://pith.science/pith/QFOVJHUN4UDBMGQYGMHDPWWHZA/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/QFOVJHUN4UDBMGQYGMHDPWWHZA/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-04T07:17:16Z","links":{"resolver":"https://pith.science/pith/QFOVJHUN4UDBMGQYGMHDPWWHZA","bundle":"https://pith.science/pith/QFOVJHUN4UDBMGQYGMHDPWWHZA/bundle.json","state":"https://pith.science/pith/QFOVJHUN4UDBMGQYGMHDPWWHZA/state.json","well_known_bundle":"https://pith.science/.well-known/pith/QFOVJHUN4UDBMGQYGMHDPWWHZA/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:QFOVJHUN4UDBMGQYGMHDPWWHZA","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"4678b2d50b7e87470347bebdbd892549fa38574b4e39ec71f26a714e03ef0c56","cross_cats_sorted":["cs.AI","cs.CL"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-16T19:03:06Z","title_canon_sha256":"cbd84d83c90de728f71e0490e56e738f645bb0471e7596d224f2f43443636bf1"},"schema_version":"1.0","source":{"id":"2506.13923","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2506.13923","created_at":"2026-07-05T11:24:23Z"},{"alias_kind":"arxiv_version","alias_value":"2506.13923v2","created_at":"2026-07-05T11:24:23Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.13923","created_at":"2026-07-05T11:24:23Z"},{"alias_kind":"pith_short_12","alias_value":"QFOVJHUN4UDB","created_at":"2026-07-05T11:24:23Z"},{"alias_kind":"pith_short_16","alias_value":"QFOVJHUN4UDBMGQY","created_at":"2026-07-05T11:24:23Z"},{"alias_kind":"pith_short_8","alias_value":"QFOVJHUN","created_at":"2026-07-05T11:24:23Z"}],"graph_snapshots":[{"event_id":"sha256:a7949d7cd4580a990de0a6cac0c35bf6531ee008999cc41a21deb7bec06235b3","target":"graph","created_at":"2026-07-05T11:24:23Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2506.13923/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"We study the process through which reasoning models trained with reinforcement learning on verifiable rewards (RLVR) can learn to solve new problems. We find that RLVR drives performance in two main ways: (1) by compressing pass@$k$ into pass@1 and (2) via \"capability gain\" in which models learn to solve new problems that they previously could not solve even at high $k$. We find that while capability gain exists across model scales, learning to solve new problems is primarily driven through self-distillation. We demonstrate these findings across model scales ranging from 0.5B to 72B parameters","authors_text":"Anisha Gunjal, Elaine Lau, Manasi Sharma, Nikhil Baharte, Sean Hendryx, Vaskar Nath","cross_cats":["cs.AI","cs.CL"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-16T19:03:06Z","title":"Adaptive Guidance Accelerates Reinforcement Learning of Reasoning Models"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.13923","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:ded41728221b80bccaeb66ebf5fdd02562a175b20177a3410b2eddb65ae4e283","target":"record","created_at":"2026-07-05T11:24:23Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"4678b2d50b7e87470347bebdbd892549fa38574b4e39ec71f26a714e03ef0c56","cross_cats_sorted":["cs.AI","cs.CL"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-16T19:03:06Z","title_canon_sha256":"cbd84d83c90de728f71e0490e56e738f645bb0471e7596d224f2f43443636bf1"},"schema_version":"1.0","source":{"id":"2506.13923","kind":"arxiv","version":2}},"canonical_sha256":"815d549e8de506161a18330e37dac7c800370fa2c87d1bcc9f7c99db0a6cb9c7","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"815d549e8de506161a18330e37dac7c800370fa2c87d1bcc9f7c99db0a6cb9c7","first_computed_at":"2026-07-05T11:24:23.973167Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:24:23.973167Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"n1mFZWVweYQw5gSNyW2NokTNYUQYOHSzsdarf8dA6Tgj1SswiGtt7XJnHXqPsZ9V4p8ZDqUzebDPeDRnmcrmBw==","signature_status":"signed_v1","signed_at":"2026-07-05T11:24:23.973773Z","signed_message":"canonical_sha256_bytes"},"source_id":"2506.13923","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:ded41728221b80bccaeb66ebf5fdd02562a175b20177a3410b2eddb65ae4e283","sha256:a7949d7cd4580a990de0a6cac0c35bf6531ee008999cc41a21deb7bec06235b3"],"state_sha256":"06b9a0010001f72ef94cc792c8f06f887273c800a10db4f2f4cc9351e6a78558"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"FAx+fXXCe5HUaHOgzK0/WkqrgIUOgoQUl54btqWtwxFmTDCP0SGWHHoRPWdwKLClw2Oqe1Blkcpa5X29ViCLCA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-04T07:17:16.084758Z","bundle_sha256":"983b53c23cdf54125a5359d85249cb8fb7138372285a02edfacf8be7c46a3ede"}}