{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:PCENNCEJFLHVMAXWTJWT6VDKX3","short_pith_number":"pith:PCENNCEJ","schema_version":"1.0","canonical_sha256":"7888d688892acf5602f69a6d3f546abeeddcadf826ca68b8a5494de05f39e926","source":{"kind":"arxiv","id":"2410.21333","version":4},"attestation_state":"computed","paper":{"title":"Mind Your Step (by Step): Chain-of-Thought can Reduce Performance on Tasks where Thinking Makes Humans Worse","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CY"],"primary_cat":"cs.LG","authors_text":"Addison J. Wu, Ilia Sucholutsky, Jiayi Geng, Ryan Liu, Tania Lombrozo, Thomas L. Griffiths","submitted_at":"2024-10-27T18:30:41Z","abstract_excerpt":"Chain-of-thought (CoT) prompting has become a widely used strategy for improving large language and multimodal model performance. However, it is still an open question under which settings CoT systematically reduces performance. In this paper, we seek to identify the characteristics of tasks where CoT reduces performance by drawing inspiration from cognitive psychology, focusing on six representative tasks from the psychological literature where deliberation hurts performance in humans. In three of these tasks, state-of-the-art models exhibit significant performance drop-offs with CoT (up to 3"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.21333","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-27T18:30:41Z","cross_cats_sorted":["cs.AI","cs.CL","cs.CY"],"title_canon_sha256":"a78a9d5cb72b9724c512911f18f0fc64b2865272ff9ca81035894f0ede449e40","abstract_canon_sha256":"13287029164340eae1c254910e26b10c4d6ae8558bf4390b22465cd3c867a433"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:21:14.794567Z","signature_b64":"Oidg/C4zZEkEtaKvX8X4Rm1VLbdM/DGBrtIOP/KEaYuKBaTWZdbf2fBzE8wSYKinDv2bnCdntSf/nGyqHCUtDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7888d688892acf5602f69a6d3f546abeeddcadf826ca68b8a5494de05f39e926","last_reissued_at":"2026-07-05T11:21:14.794094Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:21:14.794094Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Mind Your Step (by Step): Chain-of-Thought can Reduce Performance on Tasks where Thinking Makes Humans Worse","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CY"],"primary_cat":"cs.LG","authors_text":"Addison J. Wu, Ilia Sucholutsky, Jiayi Geng, Ryan Liu, Tania Lombrozo, Thomas L. Griffiths","submitted_at":"2024-10-27T18:30:41Z","abstract_excerpt":"Chain-of-thought (CoT) prompting has become a widely used strategy for improving large language and multimodal model performance. However, it is still an open question under which settings CoT systematically reduces performance. In this paper, we seek to identify the characteristics of tasks where CoT reduces performance by drawing inspiration from cognitive psychology, focusing on six representative tasks from the psychological literature where deliberation hurts performance in humans. In three of these tasks, state-of-the-art models exhibit significant performance drop-offs with CoT (up to 3"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.21333","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.21333/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.21333","created_at":"2026-07-05T11:21:14.794151+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.21333v4","created_at":"2026-07-05T11:21:14.794151+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.21333","created_at":"2026-07-05T11:21:14.794151+00:00"},{"alias_kind":"pith_short_12","alias_value":"PCENNCEJFLHV","created_at":"2026-07-05T11:21:14.794151+00:00"},{"alias_kind":"pith_short_16","alias_value":"PCENNCEJFLHVMAXW","created_at":"2026-07-05T11:21:14.794151+00:00"},{"alias_kind":"pith_short_8","alias_value":"PCENNCEJ","created_at":"2026-07-05T11:21:14.794151+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":19,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22496","citing_title":"Enabling Cloud-Level Accuracy in Edge AI through IoT Data Preprocessing","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11445","citing_title":"Forecasting Future Behavior as a Learning Task","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14084","citing_title":"CRANE: Constrained Reasoning Injection for Code Agents via Nullspace Editing","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28123","citing_title":"Risk-aware Selective Prompting for Hallucination Mitigation in Large Vision-Language Models","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22873","citing_title":"When Do LLMs Reason? A Dynamical Systems View via Entropy Phase Transitions","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2509.23108","citing_title":"Artificial Phantasia: Emergent Mental Imagery in Large Language Models","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21408","citing_title":"TCARD: Nearly Balanced Two-Level Designs with Treatment Cardinality Constraints with an Application to LLM Prompt Engineering","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10027","citing_title":"SinkTrack: Attention Sink based Context Anchoring for Large Language Models","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2508.04691","citing_title":"Before Humans Join the Team: Diagnosing Coordination Failures in Healthcare Robot Team Simulation","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2509.11295","citing_title":"The Prompt Engineering Report Distilled: Quick Start Guide for Life Sciences","ref_index":84,"is_internal_anchor":false},{"citing_arxiv_id":"2509.11295","citing_title":"The Prompt Engineering Report Distilled: Quick Start Guide for Life Sciences","ref_index":85,"is_internal_anchor":false},{"citing_arxiv_id":"2601.06993","citing_title":"Can Textual Reasoning Improve the Performance of MLLMs on Fine-grained Visual Classification?","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14084","citing_title":"CRANE: Constrained Reasoning Injection for Code Agents via Nullspace Editing","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2503.16419","citing_title":"Stop Overthinking: A Survey on Efficient Reasoning for Large Language Models","ref_index":114,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12120","citing_title":"To Whom Do Language Models Align? Measuring Principal Hierarchies Under High-Stakes Competing Demands","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09900","citing_title":"The Gordian Knot for VLMs: Diagrammatic Knot Reasoning as a Hard Benchmark","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23734","citing_title":"Prism-Reranker: Beyond Relevance Scoring -- Jointly Producing Contributions and Evidence for Agentic Retrieval","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06165","citing_title":"Post Reasoning: Improving the Performance of Non-Thinking Models at No Cost","ref_index":190,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10027","citing_title":"SinkTrack: Attention Sink based Context Anchoring for Large Language Models","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PCENNCEJFLHVMAXWTJWT6VDKX3","json":"https://pith.science/pith/PCENNCEJFLHVMAXWTJWT6VDKX3.json","graph_json":"https://pith.science/api/pith-number/PCENNCEJFLHVMAXWTJWT6VDKX3/graph.json","events_json":"https://pith.science/api/pith-number/PCENNCEJFLHVMAXWTJWT6VDKX3/events.json","paper":"https://pith.science/paper/PCENNCEJ"},"agent_actions":{"view_html":"https://pith.science/pith/PCENNCEJFLHVMAXWTJWT6VDKX3","download_json":"https://pith.science/pith/PCENNCEJFLHVMAXWTJWT6VDKX3.json","view_paper":"https://pith.science/paper/PCENNCEJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.21333&json=true","fetch_graph":"https://pith.science/api/pith-number/PCENNCEJFLHVMAXWTJWT6VDKX3/graph.json","fetch_events":"https://pith.science/api/pith-number/PCENNCEJFLHVMAXWTJWT6VDKX3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PCENNCEJFLHVMAXWTJWT6VDKX3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PCENNCEJFLHVMAXWTJWT6VDKX3/action/storage_attestation","attest_author":"https://pith.science/pith/PCENNCEJFLHVMAXWTJWT6VDKX3/action/author_attestation","sign_citation":"https://pith.science/pith/PCENNCEJFLHVMAXWTJWT6VDKX3/action/citation_signature","submit_replication":"https://pith.science/pith/PCENNCEJFLHVMAXWTJWT6VDKX3/action/replication_record"}},"created_at":"2026-07-05T11:21:14.794151+00:00","updated_at":"2026-07-05T11:21:14.794151+00:00"}