{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:YNNPUD75L4ASNNICVPP2HYFMVK","short_pith_number":"pith:YNNPUD75","schema_version":"1.0","canonical_sha256":"c35afa0ffd5f0126b502abdfa3e0acaaa677f6705217314a19fc5c87d631d8a0","source":{"kind":"arxiv","id":"2509.00125","version":1},"attestation_state":"computed","paper":{"title":"Know When to Explore: Difficulty-Aware Certainty as a Guide for LLM Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Ang Li, Shouda Liu, Yang Zhang, Yisen Wang, Zhihang Yuan","submitted_at":"2025-08-29T08:57:54Z","abstract_excerpt":"Reinforcement Learning with Verifiable Feedback (RLVF) has become a key technique for enhancing the reasoning abilities of Large Language Models (LLMs). However, its reliance on sparse, outcome based rewards, which only indicate if a final answer is correct or not, fails to provide granular guidance on the reasoning process itself. This limitation hinders efficient learning, as the model cannot distinguish between high quality and inefficient solutions, nor can it learn effectively from different types of failures. To address this, we observe that an LLMs self-certainty often correlates with t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2509.00125","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-08-29T08:57:54Z","cross_cats_sorted":[],"title_canon_sha256":"2b8d07b5d994911afdcf71e463aefab2b8126fa339e352b4db2b46e81e4eb4f7","abstract_canon_sha256":"f969a3ab71dbd6eaed20b2c2ffd61492f98bb36267a76ceef9c6c59473784d42"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:02:09.744398Z","signature_b64":"fooLMeys5gnjRtUVNyNBZyBzItby/okboX6d+Wr/4DzI47FWYAP3LQpv7nmJfHM6C9Os9WzdMmdD4CqvrEblDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c35afa0ffd5f0126b502abdfa3e0acaaa677f6705217314a19fc5c87d631d8a0","last_reissued_at":"2026-07-05T12:02:09.743594Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:02:09.743594Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Know When to Explore: Difficulty-Aware Certainty as a Guide for LLM Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Ang Li, Shouda Liu, Yang Zhang, Yisen Wang, Zhihang Yuan","submitted_at":"2025-08-29T08:57:54Z","abstract_excerpt":"Reinforcement Learning with Verifiable Feedback (RLVF) has become a key technique for enhancing the reasoning abilities of Large Language Models (LLMs). However, its reliance on sparse, outcome based rewards, which only indicate if a final answer is correct or not, fails to provide granular guidance on the reasoning process itself. This limitation hinders efficient learning, as the model cannot distinguish between high quality and inefficient solutions, nor can it learn effectively from different types of failures. To address this, we observe that an LLMs self-certainty often correlates with t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.00125","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.00125/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2509.00125","created_at":"2026-07-05T12:02:09.743707+00:00"},{"alias_kind":"arxiv_version","alias_value":"2509.00125v1","created_at":"2026-07-05T12:02:09.743707+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.00125","created_at":"2026-07-05T12:02:09.743707+00:00"},{"alias_kind":"pith_short_12","alias_value":"YNNPUD75L4AS","created_at":"2026-07-05T12:02:09.743707+00:00"},{"alias_kind":"pith_short_16","alias_value":"YNNPUD75L4ASNNIC","created_at":"2026-07-05T12:02:09.743707+00:00"},{"alias_kind":"pith_short_8","alias_value":"YNNPUD75","created_at":"2026-07-05T12:02:09.743707+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21943","citing_title":"Modularized Reinforcement Learning on LLMs: From MDP Creation to Exploration and Learning","ref_index":102,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20967","citing_title":"Formalizing Task-Space Complexity for Zero-Shot Generalization","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18089","citing_title":"From Reasoning Traces to Reusable Modules: Understanding Compositional Generalization in Language Model Reasoning","ref_index":147,"is_internal_anchor":false},{"citing_arxiv_id":"2602.07832","citing_title":"rePIRL: Learn PRM with Inverse RL for LLM Reasoning","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11328","citing_title":"Epistemic Uncertainty for Test-Time Discovery","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08283","citing_title":"HTPO: Towards Exploration-Exploitation Balanced Policy Optimization via Hierarchical Token-level Objective Control","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YNNPUD75L4ASNNICVPP2HYFMVK","json":"https://pith.science/pith/YNNPUD75L4ASNNICVPP2HYFMVK.json","graph_json":"https://pith.science/api/pith-number/YNNPUD75L4ASNNICVPP2HYFMVK/graph.json","events_json":"https://pith.science/api/pith-number/YNNPUD75L4ASNNICVPP2HYFMVK/events.json","paper":"https://pith.science/paper/YNNPUD75"},"agent_actions":{"view_html":"https://pith.science/pith/YNNPUD75L4ASNNICVPP2HYFMVK","download_json":"https://pith.science/pith/YNNPUD75L4ASNNICVPP2HYFMVK.json","view_paper":"https://pith.science/paper/YNNPUD75","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2509.00125&json=true","fetch_graph":"https://pith.science/api/pith-number/YNNPUD75L4ASNNICVPP2HYFMVK/graph.json","fetch_events":"https://pith.science/api/pith-number/YNNPUD75L4ASNNICVPP2HYFMVK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YNNPUD75L4ASNNICVPP2HYFMVK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YNNPUD75L4ASNNICVPP2HYFMVK/action/storage_attestation","attest_author":"https://pith.science/pith/YNNPUD75L4ASNNICVPP2HYFMVK/action/author_attestation","sign_citation":"https://pith.science/pith/YNNPUD75L4ASNNICVPP2HYFMVK/action/citation_signature","submit_replication":"https://pith.science/pith/YNNPUD75L4ASNNICVPP2HYFMVK/action/replication_record"}},"created_at":"2026-07-05T12:02:09.743707+00:00","updated_at":"2026-07-05T12:02:09.743707+00:00"}