{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:OQFWU423F2F3WWLG6K2YE6HP3J","short_pith_number":"pith:OQFWU423","schema_version":"1.0","canonical_sha256":"740b6a735b2e8bbb5966f2b58278efda40cae8ba288c8f07c70f6855eec32a67","source":{"kind":"arxiv","id":"2505.18298","version":1},"attestation_state":"computed","paper":{"title":"Thinking Fast and Right: Balancing Accuracy and Reasoning Length with Adaptive Rewards","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Claire Cardie, Jinyan Su","submitted_at":"2025-05-23T18:44:46Z","abstract_excerpt":"Large language models (LLMs) have demonstrated strong reasoning abilities in mathematical tasks, often enhanced through reinforcement learning (RL). However, RL-trained models frequently produce unnecessarily long reasoning traces -- even for simple queries -- leading to increased inference costs and latency. While recent approaches attempt to control verbosity by adding length penalties to the reward function, these methods rely on fixed penalty terms that are hard to tune and cannot adapt as the model's reasoning capability evolves, limiting their effectiveness. In this work, we propose an a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.18298","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-23T18:44:46Z","cross_cats_sorted":[],"title_canon_sha256":"4570f8c518a9df66e3391e080ffa90cf0557e6535cdc7d7184fc6d7ef843596c","abstract_canon_sha256":"6991b292e0848e2fad1f74a3aef76d41df6ff51c2ea1f6907c86580bcc323ec6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:08:44.354302Z","signature_b64":"T0T5qMrAZfAiAj6jF9wtzqMw3bubvC8XWkEmzBumo/rgVfPLzPloLlVFOxG3pOeePJg1G6PQ2EWT7iQDS5dOCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"740b6a735b2e8bbb5966f2b58278efda40cae8ba288c8f07c70f6855eec32a67","last_reissued_at":"2026-07-05T11:08:44.353850Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:08:44.353850Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Thinking Fast and Right: Balancing Accuracy and Reasoning Length with Adaptive Rewards","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Claire Cardie, Jinyan Su","submitted_at":"2025-05-23T18:44:46Z","abstract_excerpt":"Large language models (LLMs) have demonstrated strong reasoning abilities in mathematical tasks, often enhanced through reinforcement learning (RL). However, RL-trained models frequently produce unnecessarily long reasoning traces -- even for simple queries -- leading to increased inference costs and latency. While recent approaches attempt to control verbosity by adding length penalties to the reward function, these methods rely on fixed penalty terms that are hard to tune and cannot adapt as the model's reasoning capability evolves, limiting their effectiveness. In this work, we propose an a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.18298","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.18298/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.18298","created_at":"2026-07-05T11:08:44.353922+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.18298v1","created_at":"2026-07-05T11:08:44.353922+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.18298","created_at":"2026-07-05T11:08:44.353922+00:00"},{"alias_kind":"pith_short_12","alias_value":"OQFWU423F2F3","created_at":"2026-07-05T11:08:44.353922+00:00"},{"alias_kind":"pith_short_16","alias_value":"OQFWU423F2F3WWLG","created_at":"2026-07-05T11:08:44.353922+00:00"},{"alias_kind":"pith_short_8","alias_value":"OQFWU423","created_at":"2026-07-05T11:08:44.353922+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22716","citing_title":"Beyond Penalizing Mistakes: Stabilizing Efficiency Training in Large Reasoning Models via Adaptive Correct-Only Rewards","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19927","citing_title":"CARE: Competence-Aware Reward Shaping for Adaptive Reasoning Length in Video-MLLMs","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22211","citing_title":"CLORE: Content-Level Optimization for Reasoning Efficiency","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2601.05242","citing_title":"GDPO: Group reward-Decoupled Normalization Policy Optimization for Multi-reward RL Optimization","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07316","citing_title":"Implicit Compression Regularization: Concise Reasoning via Internal Shorter Distributions in RL Post-Training","ref_index":25,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OQFWU423F2F3WWLG6K2YE6HP3J","json":"https://pith.science/pith/OQFWU423F2F3WWLG6K2YE6HP3J.json","graph_json":"https://pith.science/api/pith-number/OQFWU423F2F3WWLG6K2YE6HP3J/graph.json","events_json":"https://pith.science/api/pith-number/OQFWU423F2F3WWLG6K2YE6HP3J/events.json","paper":"https://pith.science/paper/OQFWU423"},"agent_actions":{"view_html":"https://pith.science/pith/OQFWU423F2F3WWLG6K2YE6HP3J","download_json":"https://pith.science/pith/OQFWU423F2F3WWLG6K2YE6HP3J.json","view_paper":"https://pith.science/paper/OQFWU423","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.18298&json=true","fetch_graph":"https://pith.science/api/pith-number/OQFWU423F2F3WWLG6K2YE6HP3J/graph.json","fetch_events":"https://pith.science/api/pith-number/OQFWU423F2F3WWLG6K2YE6HP3J/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OQFWU423F2F3WWLG6K2YE6HP3J/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OQFWU423F2F3WWLG6K2YE6HP3J/action/storage_attestation","attest_author":"https://pith.science/pith/OQFWU423F2F3WWLG6K2YE6HP3J/action/author_attestation","sign_citation":"https://pith.science/pith/OQFWU423F2F3WWLG6K2YE6HP3J/action/citation_signature","submit_replication":"https://pith.science/pith/OQFWU423F2F3WWLG6K2YE6HP3J/action/replication_record"}},"created_at":"2026-07-05T11:08:44.353922+00:00","updated_at":"2026-07-05T11:08:44.353922+00:00"}