{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:NBQNU4BARAAHFXZIVHIKHOAVB7","short_pith_number":"pith:NBQNU4BA","schema_version":"1.0","canonical_sha256":"6860da7020880072df28a9d0a3b8150ff27beeed2d9896cfa30897d357ca5f0f","source":{"kind":"arxiv","id":"2607.14506","version":1},"attestation_state":"computed","paper":{"title":"Non-vacuous Generalization Bounds for Reinforcement Learning with Verifiable Rewards","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Daniel Kang, Rohan Alur, Yuxuan Zhu","submitted_at":"2026-07-16T02:42:24Z","abstract_excerpt":"While reinforcement learning with verifiable rewards (RLVR) is widely used to improve the reasoning capabilities of large language models (LLMs), the generalizability of the resulting models remains poorly understood. In this work, we establish the first non-vacuous generalization bounds for parameter-efficient RLVR fine-tuning at the billion-parameter scale. Our approach adapts PAC-Bayes compression bounds to this setting, and addresses the inherent stochasticity of token generation by applying the Gumbel-max reparameterization trick. To operationalize these bounds, we propose the Progressive"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.14506","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2026-07-16T02:42:24Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"1c2e1e1a5f00cff5f27422565124b2757ec702d619c69717d61d992adc96f2dd","abstract_canon_sha256":"48589294810226169b0eb35b18fc135adf32fca2c92e4310a827d1016b99c123"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-17T01:21:15.286967Z","signature_b64":"lTQE/YrCnwLlZ1m3OZAOlAS8K1FrMXm7NEzne7V2e33csv/ZB7O8RsPNvdDOCo2dlvgrRlWQB58GW9obs43RCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6860da7020880072df28a9d0a3b8150ff27beeed2d9896cfa30897d357ca5f0f","last_reissued_at":"2026-07-17T01:21:15.286198Z","signature_status":"signed_v1","first_computed_at":"2026-07-17T01:21:15.286198Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Non-vacuous Generalization Bounds for Reinforcement Learning with Verifiable Rewards","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Daniel Kang, Rohan Alur, Yuxuan Zhu","submitted_at":"2026-07-16T02:42:24Z","abstract_excerpt":"While reinforcement learning with verifiable rewards (RLVR) is widely used to improve the reasoning capabilities of large language models (LLMs), the generalizability of the resulting models remains poorly understood. In this work, we establish the first non-vacuous generalization bounds for parameter-efficient RLVR fine-tuning at the billion-parameter scale. Our approach adapts PAC-Bayes compression bounds to this setting, and addresses the inherent stochasticity of token generation by applying the Gumbel-max reparameterization trick. To operationalize these bounds, we propose the Progressive"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.14506","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.14506/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.14506","created_at":"2026-07-17T01:21:15.286605+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.14506v1","created_at":"2026-07-17T01:21:15.286605+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.14506","created_at":"2026-07-17T01:21:15.286605+00:00"},{"alias_kind":"pith_short_12","alias_value":"NBQNU4BARAAH","created_at":"2026-07-17T01:21:15.286605+00:00"},{"alias_kind":"pith_short_16","alias_value":"NBQNU4BARAAHFXZI","created_at":"2026-07-17T01:21:15.286605+00:00"},{"alias_kind":"pith_short_8","alias_value":"NBQNU4BA","created_at":"2026-07-17T01:21:15.286605+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NBQNU4BARAAHFXZIVHIKHOAVB7","json":"https://pith.science/pith/NBQNU4BARAAHFXZIVHIKHOAVB7.json","graph_json":"https://pith.science/api/pith-number/NBQNU4BARAAHFXZIVHIKHOAVB7/graph.json","events_json":"https://pith.science/api/pith-number/NBQNU4BARAAHFXZIVHIKHOAVB7/events.json","paper":"https://pith.science/paper/NBQNU4BA"},"agent_actions":{"view_html":"https://pith.science/pith/NBQNU4BARAAHFXZIVHIKHOAVB7","download_json":"https://pith.science/pith/NBQNU4BARAAHFXZIVHIKHOAVB7.json","view_paper":"https://pith.science/paper/NBQNU4BA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.14506&json=true","fetch_graph":"https://pith.science/api/pith-number/NBQNU4BARAAHFXZIVHIKHOAVB7/graph.json","fetch_events":"https://pith.science/api/pith-number/NBQNU4BARAAHFXZIVHIKHOAVB7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NBQNU4BARAAHFXZIVHIKHOAVB7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NBQNU4BARAAHFXZIVHIKHOAVB7/action/storage_attestation","attest_author":"https://pith.science/pith/NBQNU4BARAAHFXZIVHIKHOAVB7/action/author_attestation","sign_citation":"https://pith.science/pith/NBQNU4BARAAHFXZIVHIKHOAVB7/action/citation_signature","submit_replication":"https://pith.science/pith/NBQNU4BARAAHFXZIVHIKHOAVB7/action/replication_record"}},"created_at":"2026-07-17T01:21:15.286605+00:00","updated_at":"2026-07-17T01:21:15.286605+00:00"}