{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:JAQK2NIJZWBZKWF2OVUAL22LDR","short_pith_number":"pith:JAQK2NIJ","schema_version":"1.0","canonical_sha256":"4820ad3509cd839558ba756805eb4b1c58ea84474e32ed089956007e567fac41","source":{"kind":"arxiv","id":"2505.15612","version":1},"attestation_state":"computed","paper":{"title":"Learn to Reason Efficiently with Adaptive Length-based Reward Shaping","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Junteng Liu, Junxian He, Ruochen Zhou, Wei Liu, Yiyun Deng, Yizhe Zhang, Yuntian Deng, Yuzhen Huang","submitted_at":"2025-05-21T15:03:26Z","abstract_excerpt":"Large Reasoning Models (LRMs) have shown remarkable capabilities in solving complex problems through reinforcement learning (RL), particularly by generating long reasoning traces. However, these extended outputs often exhibit substantial redundancy, which limits the efficiency of LRMs. In this paper, we investigate RL-based approaches to promote reasoning efficiency. Specifically, we first present a unified framework that formulates various efficient reasoning methods through the lens of length-based reward shaping. Building on this perspective, we propose a novel Length-bAsed StEp Reward shap"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.15612","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-05-21T15:03:26Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"81f558f6c98fc0e45867bade9ebc2e957bf4916b3062a261799e1dfba8c37569","abstract_canon_sha256":"3394b71efb7fb560c837067475af18bd98c17afecb833518f3173b9c502bc70e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:06:48.562810Z","signature_b64":"Oh2QsF0tQyS8LUM3zo4mGcHsEzhKXWc8mfB1nWHpZwMBIDmK8DeqZ18SXnDDWo7hfnhMsHucZE4gxRZ0Ow1cCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4820ad3509cd839558ba756805eb4b1c58ea84474e32ed089956007e567fac41","last_reissued_at":"2026-07-05T11:06:48.562340Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:06:48.562340Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learn to Reason Efficiently with Adaptive Length-based Reward Shaping","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Junteng Liu, Junxian He, Ruochen Zhou, Wei Liu, Yiyun Deng, Yizhe Zhang, Yuntian Deng, Yuzhen Huang","submitted_at":"2025-05-21T15:03:26Z","abstract_excerpt":"Large Reasoning Models (LRMs) have shown remarkable capabilities in solving complex problems through reinforcement learning (RL), particularly by generating long reasoning traces. However, these extended outputs often exhibit substantial redundancy, which limits the efficiency of LRMs. In this paper, we investigate RL-based approaches to promote reasoning efficiency. Specifically, we first present a unified framework that formulates various efficient reasoning methods through the lens of length-based reward shaping. Building on this perspective, we propose a novel Length-bAsed StEp Reward shap"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.15612","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.15612/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.15612","created_at":"2026-07-05T11:06:48.562399+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.15612v1","created_at":"2026-07-05T11:06:48.562399+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.15612","created_at":"2026-07-05T11:06:48.562399+00:00"},{"alias_kind":"pith_short_12","alias_value":"JAQK2NIJZWBZ","created_at":"2026-07-05T11:06:48.562399+00:00"},{"alias_kind":"pith_short_16","alias_value":"JAQK2NIJZWBZKWF2","created_at":"2026-07-05T11:06:48.562399+00:00"},{"alias_kind":"pith_short_8","alias_value":"JAQK2NIJ","created_at":"2026-07-05T11:06:48.562399+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22716","citing_title":"Beyond Penalizing Mistakes: Stabilizing Efficiency Training in Large Reasoning Models via Adaptive Correct-Only Rewards","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19927","citing_title":"CARE: Competence-Aware Reward Shaping for Adaptive Reasoning Length in Video-MLLMs","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25604","citing_title":"DVAO: Dynamic Variance-adaptive Advantage Optimization for Multi-reward Reinforcement Learning","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30832","citing_title":"SLAT: Segment-Level Adaptive Trimming for Efficient CoT Reasoning","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22211","citing_title":"CLORE: Content-Level Optimization for Reasoning Efficiency","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08401","citing_title":"AIPO: Learning to Reason from Active Interaction","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2601.05242","citing_title":"GDPO: Group reward-Decoupled Normalization Policy Optimization for Multi-reward RL Optimization","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2503.16419","citing_title":"Stop Overthinking: A Survey on Efficient Reasoning for Large Language Models","ref_index":116,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08401","citing_title":"AIPO: Learning to Reason from Active Interaction","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09806","citing_title":"LEAD: Length-Efficient Adaptive and Dynamic Reasoning for Large Language Models","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06165","citing_title":"Post Reasoning: Improving the Performance of Non-Thinking Models at No Cost","ref_index":244,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07316","citing_title":"Implicit Compression Regularization: Concise Reasoning via Internal Shorter Distributions in RL Post-Training","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02178","citing_title":"T$^2$PO: Uncertainty-Guided Exploration Control for Stable Multi-Turn Agentic Reinforcement Learning","ref_index":11,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JAQK2NIJZWBZKWF2OVUAL22LDR","json":"https://pith.science/pith/JAQK2NIJZWBZKWF2OVUAL22LDR.json","graph_json":"https://pith.science/api/pith-number/JAQK2NIJZWBZKWF2OVUAL22LDR/graph.json","events_json":"https://pith.science/api/pith-number/JAQK2NIJZWBZKWF2OVUAL22LDR/events.json","paper":"https://pith.science/paper/JAQK2NIJ"},"agent_actions":{"view_html":"https://pith.science/pith/JAQK2NIJZWBZKWF2OVUAL22LDR","download_json":"https://pith.science/pith/JAQK2NIJZWBZKWF2OVUAL22LDR.json","view_paper":"https://pith.science/paper/JAQK2NIJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.15612&json=true","fetch_graph":"https://pith.science/api/pith-number/JAQK2NIJZWBZKWF2OVUAL22LDR/graph.json","fetch_events":"https://pith.science/api/pith-number/JAQK2NIJZWBZKWF2OVUAL22LDR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JAQK2NIJZWBZKWF2OVUAL22LDR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JAQK2NIJZWBZKWF2OVUAL22LDR/action/storage_attestation","attest_author":"https://pith.science/pith/JAQK2NIJZWBZKWF2OVUAL22LDR/action/author_attestation","sign_citation":"https://pith.science/pith/JAQK2NIJZWBZKWF2OVUAL22LDR/action/citation_signature","submit_replication":"https://pith.science/pith/JAQK2NIJZWBZKWF2OVUAL22LDR/action/replication_record"}},"created_at":"2026-07-05T11:06:48.562399+00:00","updated_at":"2026-07-05T11:06:48.562399+00:00"}