{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:RJI3Y26BO5QAUFML4PODRBGVS7","short_pith_number":"pith:RJI3Y26B","schema_version":"1.0","canonical_sha256":"8a51bc6bc177600a158be3dc3884d597c65b5fedf992d39836596ed7d2ecf0fe","source":{"kind":"arxiv","id":"2509.09675","version":1},"attestation_state":"computed","paper":{"title":"CDE: Curiosity-Driven Exploration for Efficient Reinforcement Learning in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Dian Yu, Dong Yu, Haitao Mi, Haolin Liu, Hongtu Zhu, Linfeng Song, Rui Liu, Runpeng Dai, Tong Zheng, Zhaopeng Tu, Zhenwen Liang","submitted_at":"2025-09-11T17:59:17Z","abstract_excerpt":"Reinforcement Learning with Verifiable Rewards (RLVR) is a powerful paradigm for enhancing the reasoning ability of Large Language Models (LLMs). Yet current RLVR methods often explore poorly, leading to premature convergence and entropy collapse. To address this challenge, we introduce Curiosity-Driven Exploration (CDE), a framework that leverages the model's own intrinsic sense of curiosity to guide exploration. We formalize curiosity with signals from both the actor and the critic: for the actor, we use perplexity over its generated response, and for the critic, we use the variance of value"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2509.09675","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-09-11T17:59:17Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"3b6e46bc8dedb7d95d09e44192df8b049c555ff3f4db6039d0774ad39370f176","abstract_canon_sha256":"3ecb76f0b22d37b88bbf44feb06f33607a2d423cf3145a5eaac095b52a413e96"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:09:40.695608Z","signature_b64":"zU1VhdJX/8AKC9gj9wM/ZjEIqZ0pSzANC1V497C6xIrsJVadBpkmxi7Ynasjen52pUcImwi9eL1f40rgD0u+DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8a51bc6bc177600a158be3dc3884d597c65b5fedf992d39836596ed7d2ecf0fe","last_reissued_at":"2026-07-05T12:09:40.695061Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:09:40.695061Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CDE: Curiosity-Driven Exploration for Efficient Reinforcement Learning in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Dian Yu, Dong Yu, Haitao Mi, Haolin Liu, Hongtu Zhu, Linfeng Song, Rui Liu, Runpeng Dai, Tong Zheng, Zhaopeng Tu, Zhenwen Liang","submitted_at":"2025-09-11T17:59:17Z","abstract_excerpt":"Reinforcement Learning with Verifiable Rewards (RLVR) is a powerful paradigm for enhancing the reasoning ability of Large Language Models (LLMs). Yet current RLVR methods often explore poorly, leading to premature convergence and entropy collapse. To address this challenge, we introduce Curiosity-Driven Exploration (CDE), a framework that leverages the model's own intrinsic sense of curiosity to guide exploration. We formalize curiosity with signals from both the actor and the critic: for the actor, we use perplexity over its generated response, and for the critic, we use the variance of value"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.09675","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.09675/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2509.09675","created_at":"2026-07-05T12:09:40.695147+00:00"},{"alias_kind":"arxiv_version","alias_value":"2509.09675v1","created_at":"2026-07-05T12:09:40.695147+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.09675","created_at":"2026-07-05T12:09:40.695147+00:00"},{"alias_kind":"pith_short_12","alias_value":"RJI3Y26BO5QA","created_at":"2026-07-05T12:09:40.695147+00:00"},{"alias_kind":"pith_short_16","alias_value":"RJI3Y26BO5QAUFML","created_at":"2026-07-05T12:09:40.695147+00:00"},{"alias_kind":"pith_short_8","alias_value":"RJI3Y26B","created_at":"2026-07-05T12:09:40.695147+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24994","citing_title":"ExTra: Exploratory Trajectory Optimization for Language Model Reinforcement Learning","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21943","citing_title":"Modularized Reinforcement Learning on LLMs: From MDP Creation to Exploration and Learning","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05800","citing_title":"SALT: When More Rollouts Don't Help in Group-Based Policy Optimization and How to Make Them Matter","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.28005","citing_title":"Kernelized Advantage Estimation: From Nonparametric Statistics to LLM Reasoning","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11328","citing_title":"Epistemic Uncertainty for Test-Time Discovery","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.28005","citing_title":"Kernelized Advantage Estimation: From Nonparametric Statistics to LLM Reasoning","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09262","citing_title":"Reinforcing Multimodal Reasoning Against Visual Degradation","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09269","citing_title":"DeltaRubric: Generative Multimodal Reward Modeling via Joint Planning and Verification","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12632","citing_title":"Calibration-Aware Policy Optimization for Reasoning LLMs","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18493","citing_title":"Too Correct to Learn: Reinforcement Learning on Saturated Reasoning Data","ref_index":43,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RJI3Y26BO5QAUFML4PODRBGVS7","json":"https://pith.science/pith/RJI3Y26BO5QAUFML4PODRBGVS7.json","graph_json":"https://pith.science/api/pith-number/RJI3Y26BO5QAUFML4PODRBGVS7/graph.json","events_json":"https://pith.science/api/pith-number/RJI3Y26BO5QAUFML4PODRBGVS7/events.json","paper":"https://pith.science/paper/RJI3Y26B"},"agent_actions":{"view_html":"https://pith.science/pith/RJI3Y26BO5QAUFML4PODRBGVS7","download_json":"https://pith.science/pith/RJI3Y26BO5QAUFML4PODRBGVS7.json","view_paper":"https://pith.science/paper/RJI3Y26B","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2509.09675&json=true","fetch_graph":"https://pith.science/api/pith-number/RJI3Y26BO5QAUFML4PODRBGVS7/graph.json","fetch_events":"https://pith.science/api/pith-number/RJI3Y26BO5QAUFML4PODRBGVS7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RJI3Y26BO5QAUFML4PODRBGVS7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RJI3Y26BO5QAUFML4PODRBGVS7/action/storage_attestation","attest_author":"https://pith.science/pith/RJI3Y26BO5QAUFML4PODRBGVS7/action/author_attestation","sign_citation":"https://pith.science/pith/RJI3Y26BO5QAUFML4PODRBGVS7/action/citation_signature","submit_replication":"https://pith.science/pith/RJI3Y26BO5QAUFML4PODRBGVS7/action/replication_record"}},"created_at":"2026-07-05T12:09:40.695147+00:00","updated_at":"2026-07-05T12:09:40.695147+00:00"}