{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:BUQQ2SEMMHOWEHOIYWW6C3A2PZ","short_pith_number":"pith:BUQQ2SEM","schema_version":"1.0","canonical_sha256":"0d210d488c61dd621dc8c5ade16c1a7e6fc52cd9ffd01f827c6de8fb3b27c25e","source":{"kind":"arxiv","id":"2608.09805","version":1},"attestation_state":"computed","paper":{"title":"Parameter Exploration for RLVR via Variational Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Iryna Gurevych, Nico Daheim, Vatsal Venkatkrishna","submitted_at":"2026-08-10T16:28:07Z","abstract_excerpt":"Exploration has been a focus of reinforcement learning research for a long time. Recently, there has been growing evidence that it is also an important ingredient in LLM reinforcement learning recipes that can significantly impact downstream performance. Many existing methods control exploration in the action-space, for example, using temperature scaling. However, these methods cannot reorder tokens but only influence the variance in the output distribution. This limits exploration and can lead to divergence or stalled training. Here, we investigate parameter-space exploration, where rollouts "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2608.09805","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-08-10T16:28:07Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"ae2ebc3834cf5f69b70249fac3599f09020f1ede8bbf18da1b29c98acf2d2032","abstract_canon_sha256":"65a980f192b267f4e3bcbcdd5a3e2d0c200e365abfb43c2e8f3b95f4aa019708"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-11T02:25:08.261223Z","signature_b64":"bly5Hya68dOyT1llU09o9EX2aatmloCCEGnDCNNbExykt2EEduZrJdUi61ePmb1ek6ezBxwxj0OSD2xBaFNABA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0d210d488c61dd621dc8c5ade16c1a7e6fc52cd9ffd01f827c6de8fb3b27c25e","last_reissued_at":"2026-08-11T02:25:08.259617Z","signature_status":"signed_v1","first_computed_at":"2026-08-11T02:25:08.259617Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Parameter Exploration for RLVR via Variational Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Iryna Gurevych, Nico Daheim, Vatsal Venkatkrishna","submitted_at":"2026-08-10T16:28:07Z","abstract_excerpt":"Exploration has been a focus of reinforcement learning research for a long time. Recently, there has been growing evidence that it is also an important ingredient in LLM reinforcement learning recipes that can significantly impact downstream performance. Many existing methods control exploration in the action-space, for example, using temperature scaling. However, these methods cannot reorder tokens but only influence the variance in the output distribution. This limits exploration and can lead to divergence or stalled training. Here, we investigate parameter-space exploration, where rollouts "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2608.09805","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2608.09805/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2608.09805","created_at":"2026-08-11T02:25:08.260273+00:00"},{"alias_kind":"arxiv_version","alias_value":"2608.09805v1","created_at":"2026-08-11T02:25:08.260273+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2608.09805","created_at":"2026-08-11T02:25:08.260273+00:00"},{"alias_kind":"pith_short_12","alias_value":"BUQQ2SEMMHOW","created_at":"2026-08-11T02:25:08.260273+00:00"},{"alias_kind":"pith_short_16","alias_value":"BUQQ2SEMMHOWEHOI","created_at":"2026-08-11T02:25:08.260273+00:00"},{"alias_kind":"pith_short_8","alias_value":"BUQQ2SEM","created_at":"2026-08-11T02:25:08.260273+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BUQQ2SEMMHOWEHOIYWW6C3A2PZ","json":"https://pith.science/pith/BUQQ2SEMMHOWEHOIYWW6C3A2PZ.json","graph_json":"https://pith.science/api/pith-number/BUQQ2SEMMHOWEHOIYWW6C3A2PZ/graph.json","events_json":"https://pith.science/api/pith-number/BUQQ2SEMMHOWEHOIYWW6C3A2PZ/events.json","paper":"https://pith.science/paper/BUQQ2SEM"},"agent_actions":{"view_html":"https://pith.science/pith/BUQQ2SEMMHOWEHOIYWW6C3A2PZ","download_json":"https://pith.science/pith/BUQQ2SEMMHOWEHOIYWW6C3A2PZ.json","view_paper":"https://pith.science/paper/BUQQ2SEM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2608.09805&json=true","fetch_graph":"https://pith.science/api/pith-number/BUQQ2SEMMHOWEHOIYWW6C3A2PZ/graph.json","fetch_events":"https://pith.science/api/pith-number/BUQQ2SEMMHOWEHOIYWW6C3A2PZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BUQQ2SEMMHOWEHOIYWW6C3A2PZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BUQQ2SEMMHOWEHOIYWW6C3A2PZ/action/storage_attestation","attest_author":"https://pith.science/pith/BUQQ2SEMMHOWEHOIYWW6C3A2PZ/action/author_attestation","sign_citation":"https://pith.science/pith/BUQQ2SEMMHOWEHOIYWW6C3A2PZ/action/citation_signature","submit_replication":"https://pith.science/pith/BUQQ2SEMMHOWEHOIYWW6C3A2PZ/action/replication_record"}},"created_at":"2026-08-11T02:25:08.260273+00:00","updated_at":"2026-08-11T02:25:08.260273+00:00"}