{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:Q35CKXYEOOJOLZHK7II3C465TW","short_pith_number":"pith:Q35CKXYE","schema_version":"1.0","canonical_sha256":"86fa255f047392e5e4eafa11b173dd9db86fc866ccf52adc6b4bee74612cdb96","source":{"kind":"arxiv","id":"2505.00127","version":1},"attestation_state":"computed","paper":{"title":"Between Underthinking and Overthinking: An Empirical Study of Reasoning Length and correctness in LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Claire Cardie, Jennifer Healey, Jinyan Su, Preslav Nakov","submitted_at":"2025-04-30T18:48:06Z","abstract_excerpt":"Large language models (LLMs) are increasingly optimized for long reasoning, under the assumption that more reasoning leads to better performance. However, emerging evidence suggests that longer responses can sometimes degrade accuracy rather than improve it. In this paper, we conduct a systematic empirical study of the relationship between reasoning length and answer correctness. We find that LLMs tend to overthink simple problems, generating unnecessarily long outputs, and underthink harder ones, failing to extend their reasoning when it is most needed. This indicates that models might misjud"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.00127","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-04-30T18:48:06Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"88a8b7ecb7141086dc9908c2fb832c713a5633309c69c3106592c0598c177308","abstract_canon_sha256":"e025997832c7afc47664dce7938e298ed65af71650a50d71c4dde185b2c10692"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:56:42.071132Z","signature_b64":"l7RsLk0/H+XV41t9ddrPWepiAPE1B1gChOtwM+X+N4ngjw8EpsULgiqbWvISzc2AcK7cWw+0WDZ5y7Eiat4WCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"86fa255f047392e5e4eafa11b173dd9db86fc866ccf52adc6b4bee74612cdb96","last_reissued_at":"2026-07-05T10:56:42.070692Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:56:42.070692Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Between Underthinking and Overthinking: An Empirical Study of Reasoning Length and correctness in LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Claire Cardie, Jennifer Healey, Jinyan Su, Preslav Nakov","submitted_at":"2025-04-30T18:48:06Z","abstract_excerpt":"Large language models (LLMs) are increasingly optimized for long reasoning, under the assumption that more reasoning leads to better performance. However, emerging evidence suggests that longer responses can sometimes degrade accuracy rather than improve it. In this paper, we conduct a systematic empirical study of the relationship between reasoning length and answer correctness. We find that LLMs tend to overthink simple problems, generating unnecessarily long outputs, and underthink harder ones, failing to extend their reasoning when it is most needed. This indicates that models might misjud"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.00127","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.00127/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.00127","created_at":"2026-07-05T10:56:42.070744+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.00127v1","created_at":"2026-07-05T10:56:42.070744+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.00127","created_at":"2026-07-05T10:56:42.070744+00:00"},{"alias_kind":"pith_short_12","alias_value":"Q35CKXYEOOJO","created_at":"2026-07-05T10:56:42.070744+00:00"},{"alias_kind":"pith_short_16","alias_value":"Q35CKXYEOOJOLZHK","created_at":"2026-07-05T10:56:42.070744+00:00"},{"alias_kind":"pith_short_8","alias_value":"Q35CKXYE","created_at":"2026-07-05T10:56:42.070744+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":27,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05861","citing_title":"Mitigating Factual Hallucination in Large Reasoning Models via Mixed-Mode Advantage Regularization","ref_index":12,"is_internal_anchor":true},{"citing_arxiv_id":"2606.23991","citing_title":"Critique of Agent Model","ref_index":75,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21972","citing_title":"Human vs Machine Mathematical Difficulty on Project Euler: An Experimental Analysis","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21943","citing_title":"Modularized Reinforcement Learning on LLMs: From MDP Creation to Exploration and Learning","ref_index":184,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19927","citing_title":"CARE: Competence-Aware Reward Shaping for Adaptive Reasoning Length in Video-MLLMs","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01571","citing_title":"Geometric Signatures of Reasoning: A Spectral Perspective on Task Hardness","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00482","citing_title":"Know When to Stop: Segment-Level Credit Assignment for Reducing Overthinking","ref_index":78,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03217","citing_title":"An Asymptotic Theory of Chain-of-Thought in In-Context Learning","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02218","citing_title":"Faster Synchronous On-Policy RL via Straggler-Aware Group Sizing","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01249","citing_title":"Trust Region On-Policy Distillation","ref_index":244,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00206","citing_title":"Quantized Reasoning Models Think They Need to Think Longer, but They Do Not","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15454","citing_title":"Reasoning Models Don't Just Think Longer, They Move Differently","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14358","citing_title":"Uncovering the Representation Geometry of Minimal Cores in Overcomplete Reasoning Traces","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25864","citing_title":"When Self-Belief Misleads: Active Label Acquisition for Reinforcement Learning with Verifiable Rewards","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17890","citing_title":"Dynamic Rollout Editing for Reducing Overthinking in RL-Trained Reasoning Models","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2510.20665","citing_title":"The Shape of Reasoning: Topological Analysis of Reasoning Traces in Large Language Models","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22138","citing_title":"Efficient Agentic Reasoning Through Self-Regulated Simulative Planning","ref_index":92,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22504","citing_title":"LACO: Adaptive Latent Communication for Collaborative Driving","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22211","citing_title":"CLORE: Content-Level Optimization for Reasoning Efficiency","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2602.08324","citing_title":"Towards Efficient Large Language Reasoning Models via Extreme-Ratio Chain-of-Thought Compression","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21260","citing_title":"On the Cost and Benefit of Chain of Thought: A Learning-Theoretic Perspective","ref_index":84,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15764","citing_title":"GRASP: Learning to Ground Social Reasoning in Multi-Person Non-Verbal Interactions","ref_index":73,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17672","citing_title":"Stop When Reasoning Converges: Semantic-Preserving Early Exit for Reasoning Models","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15454","citing_title":"Reasoning Models Don't Just Think Longer, They Move Differently","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2509.05489","citing_title":"Self-Aligned Reward: Towards Effective and Efficient Reasoners","ref_index":41,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Q35CKXYEOOJOLZHK7II3C465TW","json":"https://pith.science/pith/Q35CKXYEOOJOLZHK7II3C465TW.json","graph_json":"https://pith.science/api/pith-number/Q35CKXYEOOJOLZHK7II3C465TW/graph.json","events_json":"https://pith.science/api/pith-number/Q35CKXYEOOJOLZHK7II3C465TW/events.json","paper":"https://pith.science/paper/Q35CKXYE"},"agent_actions":{"view_html":"https://pith.science/pith/Q35CKXYEOOJOLZHK7II3C465TW","download_json":"https://pith.science/pith/Q35CKXYEOOJOLZHK7II3C465TW.json","view_paper":"https://pith.science/paper/Q35CKXYE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.00127&json=true","fetch_graph":"https://pith.science/api/pith-number/Q35CKXYEOOJOLZHK7II3C465TW/graph.json","fetch_events":"https://pith.science/api/pith-number/Q35CKXYEOOJOLZHK7II3C465TW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Q35CKXYEOOJOLZHK7II3C465TW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Q35CKXYEOOJOLZHK7II3C465TW/action/storage_attestation","attest_author":"https://pith.science/pith/Q35CKXYEOOJOLZHK7II3C465TW/action/author_attestation","sign_citation":"https://pith.science/pith/Q35CKXYEOOJOLZHK7II3C465TW/action/citation_signature","submit_replication":"https://pith.science/pith/Q35CKXYEOOJOLZHK7II3C465TW/action/replication_record"}},"created_at":"2026-07-05T10:56:42.070744+00:00","updated_at":"2026-07-05T10:56:42.070744+00:00"}