{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:T5QTFBVQQQOA33SIJNKJKYRMWC","short_pith_number":"pith:T5QTFBVQ","schema_version":"1.0","canonical_sha256":"9f613286b0841c0dee484b5495622cb0aed7a22ce610e438f549b853a18ecfa8","source":{"kind":"arxiv","id":"2506.05256","version":2},"attestation_state":"computed","paper":{"title":"Just Enough Thinking: Efficient Reasoning with Adaptive Length Penalties Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Chase Blagden, Chelsea Finn, Nathan Lile, Nick Haber, Rafael Rafailov, Sang Truong, Violet Xiang","submitted_at":"2025-06-05T17:17:05Z","abstract_excerpt":"Large reasoning models (LRMs) achieve higher performance on challenging reasoning tasks by generating more tokens at inference time, but this verbosity often wastes computation on easy problems. Existing solutions, including supervised finetuning on shorter traces, user-controlled budgets, or RL with uniform penalties, either require data curation, manual configuration, or treat all problems alike regardless of difficulty. We introduce Adaptive Length Penalty (ALP), a reinforcement learning objective tailoring generation length to per-prompt solve rate. During training, ALP monitors each promp"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.05256","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-06-05T17:17:05Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"13b456c64f9140c0ebb1027512273e9b0b025024316031c211c276df1d38b061","abstract_canon_sha256":"8a0ad8769ae18675831b519676478ceb6e5aa0b5f70ed36ded0b44f6f7ec00a0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:17:00.939928Z","signature_b64":"riREJyBc9FBAvmc2X57EXvYnxUJssf4RyogJ7ZXvnUDGwcXcBSSfk4LUC+Wx5fY8SEuYy96dD+UG80MT3+40CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9f613286b0841c0dee484b5495622cb0aed7a22ce610e438f549b853a18ecfa8","last_reissued_at":"2026-07-05T11:17:00.939439Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:17:00.939439Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Just Enough Thinking: Efficient Reasoning with Adaptive Length Penalties Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Chase Blagden, Chelsea Finn, Nathan Lile, Nick Haber, Rafael Rafailov, Sang Truong, Violet Xiang","submitted_at":"2025-06-05T17:17:05Z","abstract_excerpt":"Large reasoning models (LRMs) achieve higher performance on challenging reasoning tasks by generating more tokens at inference time, but this verbosity often wastes computation on easy problems. Existing solutions, including supervised finetuning on shorter traces, user-controlled budgets, or RL with uniform penalties, either require data curation, manual configuration, or treat all problems alike regardless of difficulty. We introduce Adaptive Length Penalty (ALP), a reinforcement learning objective tailoring generation length to per-prompt solve rate. During training, ALP monitors each promp"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.05256","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.05256/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.05256","created_at":"2026-07-05T11:17:00.939500+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.05256v2","created_at":"2026-07-05T11:17:00.939500+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.05256","created_at":"2026-07-05T11:17:00.939500+00:00"},{"alias_kind":"pith_short_12","alias_value":"T5QTFBVQQQOA","created_at":"2026-07-05T11:17:00.939500+00:00"},{"alias_kind":"pith_short_16","alias_value":"T5QTFBVQQQOA33SI","created_at":"2026-07-05T11:17:00.939500+00:00"},{"alias_kind":"pith_short_8","alias_value":"T5QTFBVQ","created_at":"2026-07-05T11:17:00.939500+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":18,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24320","citing_title":"ZONOS2 Technical Report","ref_index":268,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22716","citing_title":"Beyond Penalizing Mistakes: Stabilizing Efficiency Training in Large Reasoning Models via Adaptive Correct-Only Rewards","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21943","citing_title":"Modularized Reinforcement Learning on LLMs: From MDP Creation to Exploration and Learning","ref_index":231,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20295","citing_title":"Token-Operations-Oriented Inference Optimization Techniques for Large Models","ref_index":88,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18089","citing_title":"From Reasoning Traces to Reusable Modules: Understanding Compositional Generalization in Language Model Reasoning","ref_index":82,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03077","citing_title":"Libra: Efficient Resource Management for Agentic RL Post-Training","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01249","citing_title":"Trust Region On-Policy Distillation","ref_index":243,"is_internal_anchor":false},{"citing_arxiv_id":"2606.24320","citing_title":"ZONOS2 Technical Report","ref_index":268,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22211","citing_title":"CLORE: Content-Level Optimization for Reasoning Efficiency","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19358","citing_title":"Taming the Thinker: Conditional Entropy Shaping for Adaptive LLM Reasoning","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2512.19995","citing_title":"Schoenfeld's Anatomy of Mathematical Reasoning by Language Models","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2601.11340","citing_title":"Neural Chain-of-Thought Search: Searching the Optimal Reasoning Path to Enhance Large Language Models","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2602.09953","citing_title":"ATTNPO: Attention-Guided Process Supervision for Efficient Reasoning","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2603.08659","citing_title":"CODA: Difficulty-Aware Compute Allocation for Adaptive Reasoning","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27039","citing_title":"Length Value Model: Scalable Value Pretraining for Token-Level Length Modeling","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08441","citing_title":"DUET: Optimize Token-Budget Allocation for Reinforcement Learning with Verifiable Rewards","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09806","citing_title":"LEAD: Length-Efficient Adaptive and Dynamic Reasoning for Large Language Models","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05365","citing_title":"ZAYA1-8B Technical Report","ref_index":227,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/T5QTFBVQQQOA33SIJNKJKYRMWC","json":"https://pith.science/pith/T5QTFBVQQQOA33SIJNKJKYRMWC.json","graph_json":"https://pith.science/api/pith-number/T5QTFBVQQQOA33SIJNKJKYRMWC/graph.json","events_json":"https://pith.science/api/pith-number/T5QTFBVQQQOA33SIJNKJKYRMWC/events.json","paper":"https://pith.science/paper/T5QTFBVQ"},"agent_actions":{"view_html":"https://pith.science/pith/T5QTFBVQQQOA33SIJNKJKYRMWC","download_json":"https://pith.science/pith/T5QTFBVQQQOA33SIJNKJKYRMWC.json","view_paper":"https://pith.science/paper/T5QTFBVQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.05256&json=true","fetch_graph":"https://pith.science/api/pith-number/T5QTFBVQQQOA33SIJNKJKYRMWC/graph.json","fetch_events":"https://pith.science/api/pith-number/T5QTFBVQQQOA33SIJNKJKYRMWC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/T5QTFBVQQQOA33SIJNKJKYRMWC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/T5QTFBVQQQOA33SIJNKJKYRMWC/action/storage_attestation","attest_author":"https://pith.science/pith/T5QTFBVQQQOA33SIJNKJKYRMWC/action/author_attestation","sign_citation":"https://pith.science/pith/T5QTFBVQQQOA33SIJNKJKYRMWC/action/citation_signature","submit_replication":"https://pith.science/pith/T5QTFBVQQQOA33SIJNKJKYRMWC/action/replication_record"}},"created_at":"2026-07-05T11:17:00.939500+00:00","updated_at":"2026-07-05T11:17:00.939500+00:00"}