{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ZTCDUZCNT37NLXFCKRJW7EIMTF","short_pith_number":"pith:ZTCDUZCN","schema_version":"1.0","canonical_sha256":"ccc43a644d9efed5dca254536f910c9946cbe25433731b1d659fd73c0692d2ab","source":{"kind":"arxiv","id":"2506.06395","version":3},"attestation_state":"computed","paper":{"title":"Confidence Is All You Need: Few-Shot RL Fine-Tuning of Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Alexander Zubrey, Andrey Kuznetsov, Ivan Oseledets, Matvey Skripkin, Pengyi Li","submitted_at":"2025-06-05T19:55:15Z","abstract_excerpt":"Large language models (LLMs) excel at reasoning, yet post-training remains critical for aligning their behavior with task goals. Existing reinforcement learning (RL) methods often depend on costly human annotations or external reward models. We propose Reinforcement Learning via Self-Confidence (RLSC), which uses the model's own confidence as reward signals-eliminating the need for labels, preference models, or reward engineering. Applied to Qwen2.5-Math-7B with only 16 samples per question and 10 or 20 training steps, RLSC improves accuracy by +13.4% on AIME2024, +21.2% on MATH500, +21.7% on "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.06395","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-05T19:55:15Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"e203482ea520cc5ee9e4f59decdb773f60846f2380fb2755758da36b6310fa3d","abstract_canon_sha256":"cab533dccb6630d887a7febd84fe95d2ceff093b7a9be65bfbd086a5c46dc92b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:19:28.486029Z","signature_b64":"WtwGZCLh69qHM3sHB0ekGDDO53foLYN5rRoOyXYdz0axKMYlftQ/Jp7wqq+5/T/j2FO47Jajb6YelfUJlG4GAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ccc43a644d9efed5dca254536f910c9946cbe25433731b1d659fd73c0692d2ab","last_reissued_at":"2026-07-05T11:19:28.485486Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:19:28.485486Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Confidence Is All You Need: Few-Shot RL Fine-Tuning of Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Alexander Zubrey, Andrey Kuznetsov, Ivan Oseledets, Matvey Skripkin, Pengyi Li","submitted_at":"2025-06-05T19:55:15Z","abstract_excerpt":"Large language models (LLMs) excel at reasoning, yet post-training remains critical for aligning their behavior with task goals. Existing reinforcement learning (RL) methods often depend on costly human annotations or external reward models. We propose Reinforcement Learning via Self-Confidence (RLSC), which uses the model's own confidence as reward signals-eliminating the need for labels, preference models, or reward engineering. Applied to Qwen2.5-Math-7B with only 16 samples per question and 10 or 20 training steps, RLSC improves accuracy by +13.4% on AIME2024, +21.2% on MATH500, +21.7% on "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.06395","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.06395/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.06395","created_at":"2026-07-05T11:19:28.485548+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.06395v3","created_at":"2026-07-05T11:19:28.485548+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.06395","created_at":"2026-07-05T11:19:28.485548+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZTCDUZCNT37N","created_at":"2026-07-05T11:19:28.485548+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZTCDUZCNT37NLXFC","created_at":"2026-07-05T11:19:28.485548+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZTCDUZCN","created_at":"2026-07-05T11:19:28.485548+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":17,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20881","citing_title":"When Do Intrinsic Rewards Work for Code Reasoning? A Comprehensive Study","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11634","citing_title":"Architecture-Aware Reinforcement Learning Makes Sliding-Window Attention Competitive in Math Reasoning","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04503","citing_title":"Smart Picks in the Dark: Towards Efficient RLVR for Reasoning via Tracing Metacognitive Pivots","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04516","citing_title":"GeoMin: Data-Efficient Semi-Supervised RLVR via Geometric Distribution Modeling","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01249","citing_title":"Trust Region On-Policy Distillation","ref_index":80,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19444","citing_title":"Detecting and Mitigating the Correct-Answer Extinction Window in Test-Time Reinforcement Learning with Majority Voting","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2509.14234","citing_title":"Compute as Teacher: Turning Inference Compute Into Reference-Free Supervision","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2509.08827","citing_title":"A Survey of Reinforcement Learning for Large Reasoning Models","ref_index":280,"is_internal_anchor":false},{"citing_arxiv_id":"2603.09117","citing_title":"Decoupling Reasoning and Confidence: Resurrecting Calibration in Reinforcement Learning from Verifiable Rewards","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15012","citing_title":"Boosting Reinforcement Learning with Verifiable Rewards via Randomly Selected Few-Shot Guidance","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13467","citing_title":"PDCR: Perception-Decomposed Confidence Reward for Vision-Language Reasoning","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03993","citing_title":"Can LLMs Learn to Reason Robustly under Noisy Supervision?","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01428","citing_title":"Hallucinations Undermine Trust; Metacognition is a Way Forward","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01853","citing_title":"Spatiotemporal Hidden-State Dynamics as a Signature of Internal Reasoning in Large Language Models","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04065","citing_title":"Free Energy-Driven Reinforcement Learning with Adaptive Advantage Shaping for Unsupervised Reasoning in LLMs","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07244","citing_title":"Experience Sharing in Mutual Reinforcement Learning for Heterogeneous Language Models","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17928","citing_title":"HEALing Entropy Collapse: Enhancing Exploration in Few-Shot RLVR via Hybrid-Domain Entropy Dynamics Alignment","ref_index":34,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZTCDUZCNT37NLXFCKRJW7EIMTF","json":"https://pith.science/pith/ZTCDUZCNT37NLXFCKRJW7EIMTF.json","graph_json":"https://pith.science/api/pith-number/ZTCDUZCNT37NLXFCKRJW7EIMTF/graph.json","events_json":"https://pith.science/api/pith-number/ZTCDUZCNT37NLXFCKRJW7EIMTF/events.json","paper":"https://pith.science/paper/ZTCDUZCN"},"agent_actions":{"view_html":"https://pith.science/pith/ZTCDUZCNT37NLXFCKRJW7EIMTF","download_json":"https://pith.science/pith/ZTCDUZCNT37NLXFCKRJW7EIMTF.json","view_paper":"https://pith.science/paper/ZTCDUZCN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.06395&json=true","fetch_graph":"https://pith.science/api/pith-number/ZTCDUZCNT37NLXFCKRJW7EIMTF/graph.json","fetch_events":"https://pith.science/api/pith-number/ZTCDUZCNT37NLXFCKRJW7EIMTF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZTCDUZCNT37NLXFCKRJW7EIMTF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZTCDUZCNT37NLXFCKRJW7EIMTF/action/storage_attestation","attest_author":"https://pith.science/pith/ZTCDUZCNT37NLXFCKRJW7EIMTF/action/author_attestation","sign_citation":"https://pith.science/pith/ZTCDUZCNT37NLXFCKRJW7EIMTF/action/citation_signature","submit_replication":"https://pith.science/pith/ZTCDUZCNT37NLXFCKRJW7EIMTF/action/replication_record"}},"created_at":"2026-07-05T11:19:28.485548+00:00","updated_at":"2026-07-05T11:19:28.485548+00:00"}