{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:CPMCKUJAO2EK2ZVUPZOBGESHYG","short_pith_number":"pith:CPMCKUJA","schema_version":"1.0","canonical_sha256":"13d82551207688ad66b47e5c131247c1a14c3fe058bc9a7aac01b761ec2faaa0","source":{"kind":"arxiv","id":"2502.08235","version":1},"attestation_state":"computed","paper":{"title":"The Danger of Overthinking: Examining the Reasoning-Action Dilemma in Agentic Tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Aditya Desai, Alejandro Cuadron, Ana Klimovic, Dacheng Li, Graham Neubig, Huanzhi Mao, Ion Stoica, Joseph E. Gonzalez, Luis Gaspar Schroeder, Nicholas Thumiger, Shu Liu, Siyuan Zhuang, Tian Xia, Wenjie Ma, Xingyao Wang, Yichuan Wang","submitted_at":"2025-02-12T09:23:26Z","abstract_excerpt":"Large Reasoning Models (LRMs) represent a breakthrough in AI problem-solving capabilities, but their effectiveness in interactive environments can be limited. This paper introduces and analyzes overthinking in LRMs. A phenomenon where models favor extended internal reasoning chains over environmental interaction. Through experiments on software engineering tasks using SWE Bench Verified, we observe three recurring patterns: Analysis Paralysis, Rogue Actions, and Premature Disengagement. We propose a framework to study these behaviors, which correlates with human expert assessments, and analyze"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.08235","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-02-12T09:23:26Z","cross_cats_sorted":[],"title_canon_sha256":"88fe8bab2d5a1e21bd3db23fcfd8024e4d08fa080b9edfa858f7407daa473093","abstract_canon_sha256":"e939a5a0871fc09a18f311b41dcfbd9ef9688a000186446ec88389dd61396b13"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:13:19.821509Z","signature_b64":"UR/bNvDRmo58ysBnGvaFSJ6YdXxq06DB6LVECHvzf0YP5bQxtnDYW5B+Ekiu227HwmRsuQ+VC5gknuyRyuUyCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"13d82551207688ad66b47e5c131247c1a14c3fe058bc9a7aac01b761ec2faaa0","last_reissued_at":"2026-07-05T10:13:19.821014Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:13:19.821014Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Danger of Overthinking: Examining the Reasoning-Action Dilemma in Agentic Tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Aditya Desai, Alejandro Cuadron, Ana Klimovic, Dacheng Li, Graham Neubig, Huanzhi Mao, Ion Stoica, Joseph E. Gonzalez, Luis Gaspar Schroeder, Nicholas Thumiger, Shu Liu, Siyuan Zhuang, Tian Xia, Wenjie Ma, Xingyao Wang, Yichuan Wang","submitted_at":"2025-02-12T09:23:26Z","abstract_excerpt":"Large Reasoning Models (LRMs) represent a breakthrough in AI problem-solving capabilities, but their effectiveness in interactive environments can be limited. This paper introduces and analyzes overthinking in LRMs. A phenomenon where models favor extended internal reasoning chains over environmental interaction. Through experiments on software engineering tasks using SWE Bench Verified, we observe three recurring patterns: Analysis Paralysis, Rogue Actions, and Premature Disengagement. We propose a framework to study these behaviors, which correlates with human expert assessments, and analyze"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.08235","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.08235/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.08235","created_at":"2026-07-05T10:13:19.821074+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.08235v1","created_at":"2026-07-05T10:13:19.821074+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.08235","created_at":"2026-07-05T10:13:19.821074+00:00"},{"alias_kind":"pith_short_12","alias_value":"CPMCKUJAO2EK","created_at":"2026-07-05T10:13:19.821074+00:00"},{"alias_kind":"pith_short_16","alias_value":"CPMCKUJAO2EK2ZVU","created_at":"2026-07-05T10:13:19.821074+00:00"},{"alias_kind":"pith_short_8","alias_value":"CPMCKUJA","created_at":"2026-07-05T10:13:19.821074+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":20,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23537","citing_title":"SQLConductor: Search-to-Policy Learning for Step-wise Text-to-SQL Orchestration","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11470","citing_title":"The Periodic Table of LLM Reasoning: A Structured Survey of Reasoning Paradigms, Methods, and Failure Modes","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05391","citing_title":"Human oversight of agentic systems in practice: Examining the oversight work, challenges, and heuristics of developers using software agents","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02835","citing_title":"Thinking Past the Answer: Evaluating Harmful Overthinking in Large Reasoning Models","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02411","citing_title":"FitText: Evolving Agent Tool Ecologies via Memetic Retrieval","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17890","citing_title":"Dynamic Rollout Editing for Reducing Overthinking in RL-Trained Reasoning Models","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2509.06337","citing_title":"Large Language Models as Virtual Survey Respondents: Evaluating Sociodemographic Response Generation","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2509.05489","citing_title":"Self-Aligned Reward: Towards Effective and Efficient Reasoners","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13338","citing_title":"Inducing Overthink: Hierarchical Genetic Algorithm-based DoS Attack on Black-Box Large Language Reasoning Models","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13414","citing_title":"TRIAGE: Evaluating Prospective Metacognitive Control in LLMs under Resource Constraints","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13338","citing_title":"Inducing Overthink: Hierarchical Genetic Algorithm-based DoS Attack on Black-Box Large Language Reasoning Models","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04009","citing_title":"Benchmarking and Evaluating VLMs for Software Architecture Diagram Understanding","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2503.09567","citing_title":"Towards Reasoning Era: A Survey of Long Chain-of-Thought for Reasoning Large Language Models","ref_index":145,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04066","citing_title":"Adapt to Thrive! Adaptive Power-Mean Policy Optimization for Improved LLM Reasoning","ref_index":249,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04065","citing_title":"Free Energy-Driven Reinforcement Learning with Adaptive Advantage Shaping for Unsupervised Reasoning in LLMs","ref_index":264,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06787","citing_title":"When Is Thinking Enough? Early Exit via Sufficiency Assessment for Efficient Reasoning","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07316","citing_title":"Implicit Compression Regularization: Concise Reasoning via Internal Shorter Distributions in RL Post-Training","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17304","citing_title":"Efficient Test-Time Scaling via Temporal Reasoning Aggregation","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02411","citing_title":"FitText: Evolving Agent Tool Ecologies via Memetic Retrieval","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02661","citing_title":"AcademiClaw: When Students Set Challenges for AI Agents","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CPMCKUJAO2EK2ZVUPZOBGESHYG","json":"https://pith.science/pith/CPMCKUJAO2EK2ZVUPZOBGESHYG.json","graph_json":"https://pith.science/api/pith-number/CPMCKUJAO2EK2ZVUPZOBGESHYG/graph.json","events_json":"https://pith.science/api/pith-number/CPMCKUJAO2EK2ZVUPZOBGESHYG/events.json","paper":"https://pith.science/paper/CPMCKUJA"},"agent_actions":{"view_html":"https://pith.science/pith/CPMCKUJAO2EK2ZVUPZOBGESHYG","download_json":"https://pith.science/pith/CPMCKUJAO2EK2ZVUPZOBGESHYG.json","view_paper":"https://pith.science/paper/CPMCKUJA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.08235&json=true","fetch_graph":"https://pith.science/api/pith-number/CPMCKUJAO2EK2ZVUPZOBGESHYG/graph.json","fetch_events":"https://pith.science/api/pith-number/CPMCKUJAO2EK2ZVUPZOBGESHYG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CPMCKUJAO2EK2ZVUPZOBGESHYG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CPMCKUJAO2EK2ZVUPZOBGESHYG/action/storage_attestation","attest_author":"https://pith.science/pith/CPMCKUJAO2EK2ZVUPZOBGESHYG/action/author_attestation","sign_citation":"https://pith.science/pith/CPMCKUJAO2EK2ZVUPZOBGESHYG/action/citation_signature","submit_replication":"https://pith.science/pith/CPMCKUJAO2EK2ZVUPZOBGESHYG/action/replication_record"}},"created_at":"2026-07-05T10:13:19.821074+00:00","updated_at":"2026-07-05T10:13:19.821074+00:00"}