{"work":{"id":"8c6d5a6b-b5cc-4105-9c84-9c34bb9375bb","openalex_id":"https://openalex.org/W4409201831","doi":"10.1101/2025.04.03.646459","arxiv_id":"2203.11171","raw_key":null,"title":"Self-Consistency Improves Chain of Thought Reasoning in Language Models","authors":null,"authors_text":"Xuezhi Wang, Jason Wei, Dale Schuurmans, Quoc Le, Ed Chi, Sharan Narang, Aakanksha Chowdhery, and Denny Zhou","year":2022,"venue":"cs.CL","abstract":"Chain-of-thought prompting combined with pre-trained large language models has achieved encouraging results on complex reasoning tasks. In this paper, we propose a new decoding strategy, self-consistency, to replace the naive greedy decoding used in chain-of-thought prompting. It first samples a diverse set of reasoning paths instead of only taking the greedy one, and then selects the most consistent answer by marginalizing out the sampled reasoning paths. Self-consistency leverages the intuition that a complex reasoning problem typically admits multiple different ways of thinking leading to its unique correct answer. Our extensive empirical evaluation shows that self-consistency boosts the performance of chain-of-thought prompting with a striking margin on a range of popular arithmetic and commonsense reasoning benchmarks, including GSM8K (+17.9%), SVAMP (+11.0%), AQuA (+12.2%), StrategyQA (+6.4%) and ARC-challenge (+3.9%).","external_url":"https://arxiv.org/abs/2203.11171","cited_by_count":42,"metadata_source":"pith","metadata_fetched_at":"2026-08-05T02:28:24.338817+00:00","pith_arxiv_id":"2203.11171","created_at":"2026-05-09T05:50:25.673732+00:00","updated_at":"2026-08-05T02:28:24.338817+00:00","title_quality_ok":true,"display_title":"Self-Consistency Improves Chain of Thought Reasoning in Language Models","render_title":"Self-Consistency Improves Chain of Thought Reasoning in Language Models"},"hub":{"state":{"work_id":"8c6d5a6b-b5cc-4105-9c84-9c34bb9375bb","tier":"super_hub","tier_reason":"100+ Pith inbound or 10,000+ external citations","pith_inbound_count":458,"external_cited_by_count":42,"distinct_field_count":28,"first_pith_cited_at":"2022-01-28T02:33:07+00:00","last_pith_cited_at":"2026-07-09T16:34:15+00:00","author_build_status":"needed","summary_status":"needed","contexts_status":"needed","graph_status":"needed","ask_index_status":"needed","reader_status":"not_needed","recognition_status":"not_needed","updated_at":"2026-08-22T16:09:33.231320+00:00","tier_text":"super_hub"},"tier":"super_hub","role_counts":[{"context_role":"background","n":68},{"context_role":"method","n":8},{"context_role":"baseline","n":3},{"context_role":"dataset","n":1},{"context_role":"extension","n":1}],"polarity_counts":[{"context_polarity":"background","n":62},{"context_polarity":"use_method","n":8},{"context_polarity":"baseline","n":3},{"context_polarity":"support","n":3},{"context_polarity":"unclear","n":3},{"context_polarity":"extend","n":1},{"context_polarity":"use_dataset","n":1}],"runs":{"ask_index":{"job_type":"ask_index","status":"succeeded","result":{"title":"Self-Consistency Improves Chain of Thought Reasoning in Language Models","claims":[{"claim_text":"Chain-of-thought prompting combined with pre-trained large language models has achieved encouraging results on complex reasoning tasks. In this paper, we propose a new decoding strategy, self-consistency, to replace the naive greedy decoding used in chain-of-thought prompting. It first samples a diverse set of reasoning paths instead of only taking the greedy one, and then selects the most consistent answer by marginalizing out the sampled reasoning paths. Self-consistency leverages the intuition that a complex reasoning problem typically admits multiple different ways of thinking leading to i","claim_type":"abstract","evidence_strength":"source_metadata"}],"why_cited":"Pith tracks Self-Consistency Improves Chain of Thought Reasoning in Language Models because it crossed a citation-hub threshold.","role_counts":[]},"error":null,"updated_at":"2026-05-13T22:23:43.070898+00:00"},"author_expand":{"job_type":"author_expand","status":"succeeded","result":{"authors_linked":[{"id":"7d44fff2-95d1-41fc-83c5-2969ba35b464","orcid":null,"display_name":"Xuezhi Wang"},{"id":"cb0687c7-02a4-474c-a3bd-ba5a6e0d00f9","orcid":null,"display_name":"Jason Wei"},{"id":"c5de7f09-6e32-4784-8fa7-9532f012735a","orcid":null,"display_name":"Dale Schuurmans"},{"id":"b7d93d8f-b3b6-4571-b881-819e8f1bae2d","orcid":null,"display_name":"Quoc Le"},{"id":"46959ee3-5c32-49cb-becf-e7baf14d04f2","orcid":null,"display_name":"Ed Chi"},{"id":"e0e3c462-ade2-43d1-974e-1a346b19be3b","orcid":null,"display_name":"Sharan Narang"}]},"error":null,"updated_at":"2026-05-13T22:23:43.666609+00:00"},"context_extract":{"job_type":"context_extract","status":"succeeded","result":{"enqueued_papers":25},"error":null,"updated_at":"2026-05-13T22:23:45.324149+00:00"},"graph_features":{"job_type":"graph_features","status":"succeeded","result":{"co_cited":[{"title":"Training Verifiers to Solve Math Word Problems","work_id":"acab1aa8-b4d6-40e0-a3ee-25341701dca2","shared_citers":40},{"title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","work_id":"e6b75ad5-2877-4168-97c8-710407094d20","shared_citers":36},{"title":"Qwen3 Technical Report","work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e","shared_citers":28},{"title":"Chain-of-Thought Prompting Elicits Reasoning in Large Language Models","work_id":"d1cf6693-a082-403c-ada9-dac7b96341f9","shared_citers":27},{"title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","work_id":"c5006563-f3ec-438a-9e35-b7b484f34828","shared_citers":25},{"title":"GPT-4 Technical Report","work_id":"b928e041-6991-4c08-8c81-0359e4097c7b","shared_citers":22},{"title":"Least-to-Most Prompting Enables Complex Reasoning in Large Language Models","work_id":"7e58c111-4666-4996-b5ad-1c8efd433083","shared_citers":22},{"title":"Measuring Mathematical Problem Solving With the MATH Dataset","work_id":"50652ac6-fb7c-4675-a2c2-159c241feb17","shared_citers":22},{"title":"Evaluating Large Language Models Trained on Code","work_id":"042493e9-b26f-4b4e-bbde-382072ca9b08","shared_citers":21},{"title":"ReAct: Synergizing Reasoning and Acting in Language Models","work_id":"407a2351-25f1-497d-b611-f77d0292a8e6","shared_citers":20},{"title":"Scaling LLM Test-Time Compute Optimally can be More Effective than Scaling Model Parameters","work_id":"a8d50b24-bdf5-46ed-bc4f-2927dfd81f1d","shared_citers":19},{"title":"The Llama 3 Herd of Models","work_id":"1549a635-88af-4ac1-acfe-51ae7bb53345","shared_citers":16},{"title":"Self-Refine: Iterative Refinement with Self-Feedback","work_id":"59181e7f-e58e-45d3-8146-4477a9f53d5a","shared_citers":15},{"title":"Language Models (Mostly) Know What They Know","work_id":"8ca58a10-da41-4f70-baae-7e449512e345","shared_citers":14},{"title":"LLaMA: Open and Efficient Foundation Language Models","work_id":"c018fc23-6f3f-4035-9d02-28a2173b2b9d","shared_citers":14},{"title":"Large Language Models are Zero-Shot Reasoners","work_id":"d9b7eb1a-7165-46ff-9f06-d2f0b9d6f95d","shared_citers":13},{"title":"Qwen2.5 Technical Report","work_id":"d8432992-4980-4a81-85c7-9fa2c2b87f85","shared_citers":13},{"title":"Scaling Laws for Neural Language Models","work_id":"b7dd8749-9c45-4977-ab9b-64478dce1ae8","shared_citers":13},{"title":"Training language models to follow instructions with human feedback","work_id":"52aff42f-4fa9-4fcf-bdb3-1459b9bebf65","shared_citers":13},{"title":"DAPO: An Open-Source LLM Reinforcement Learning System at Scale","work_id":"64019d00-0b11-4bbd-b173-b46c8fad0157","shared_citers":12},{"title":"Reflexion: Language Agents with Verbal Reinforcement Learning","work_id":"778f739e-5f55-4961-8a2a-e4736a2757f4","shared_citers":12},{"title":"OpenAI o1 System Card","work_id":"68d3c334-0fc9-49e3-b7b0-a69afae933e2","shared_citers":11},{"title":"Program of Thoughts Prompting: Disentangling Computation from Reasoning for Numerical Reasoning Tasks","work_id":"618aa44c-a6c6-425c-abce-8aa8aa842921","shared_citers":11},{"title":"Proximal Policy Optimization Algorithms","work_id":"240c67fe-d14d-4520-91c1-38a4e272ca19","shared_citers":11}],"time_series":[{"n":4,"year":2022},{"n":5,"year":2023},{"n":6,"year":2024},{"n":1,"year":2025},{"n":136,"year":2026}]},"error":null,"updated_at":"2026-05-13T22:23:35.942320+00:00"},"identity_refresh":{"job_type":"identity_refresh","status":"succeeded","result":{"fixed":1,"items":[{"title":"Qwen3 Technical Report","work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e","resolver":"local_arxiv","confidence":0.98,"old_work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e"}],"errors":[],"attempted":1},"error":null,"updated_at":"2026-05-13T22:23:40.013155+00:00"},"role_polarity":{"job_type":"role_polarity","status":"succeeded","result":{"title":"Self-Consistency Improves Chain of Thought Reasoning in Language Models","claims":[{"claim_text":"Chain-of-thought prompting combined with pre-trained large language models has achieved encouraging results on complex reasoning tasks. In this paper, we propose a new decoding strategy, self-consistency, to replace the naive greedy decoding used in chain-of-thought prompting. It first samples a diverse set of reasoning paths instead of only taking the greedy one, and then selects the most consistent answer by marginalizing out the sampled reasoning paths. Self-consistency leverages the intuition that a complex reasoning problem typically admits multiple different ways of thinking leading to i","claim_type":"abstract","evidence_strength":"source_metadata"}],"why_cited":"Pith tracks Self-Consistency Improves Chain of Thought Reasoning in Language Models because it crossed a citation-hub threshold.","role_counts":[]},"error":null,"updated_at":"2026-05-13T22:23:45.326243+00:00"},"summary_claims":{"job_type":"summary_claims","status":"succeeded","result":{"title":"Self-Consistency Improves Chain of Thought Reasoning in Language Models","claims":[{"claim_text":"Chain-of-thought prompting combined with pre-trained large language models has achieved encouraging results on complex reasoning tasks. In this paper, we propose a new decoding strategy, self-consistency, to replace the naive greedy decoding used in chain-of-thought prompting. It first samples a diverse set of reasoning paths instead of only taking the greedy one, and then selects the most consistent answer by marginalizing out the sampled reasoning paths. Self-consistency leverages the intuition that a complex reasoning problem typically admits multiple different ways of thinking leading to i","claim_type":"abstract","evidence_strength":"source_metadata"}],"why_cited":"Pith tracks Self-Consistency Improves Chain of Thought Reasoning in Language Models because it crossed a citation-hub threshold.","role_counts":[]},"error":null,"updated_at":"2026-05-13T22:23:43.069149+00:00"}},"summary":{"title":"Self-Consistency Improves Chain of Thought Reasoning in Language Models","claims":[{"claim_text":"Chain-of-thought prompting combined with pre-trained large language models has achieved encouraging results on complex reasoning tasks. In this paper, we propose a new decoding strategy, self-consistency, to replace the naive greedy decoding used in chain-of-thought prompting. It first samples a diverse set of reasoning paths instead of only taking the greedy one, and then selects the most consistent answer by marginalizing out the sampled reasoning paths. Self-consistency leverages the intuition that a complex reasoning problem typically admits multiple different ways of thinking leading to i","claim_type":"abstract","evidence_strength":"source_metadata"}],"why_cited":"Pith tracks Self-Consistency Improves Chain of Thought Reasoning in Language Models because it crossed a citation-hub threshold.","role_counts":[]},"graph":{"co_cited":[{"title":"Training Verifiers to Solve Math Word Problems","work_id":"acab1aa8-b4d6-40e0-a3ee-25341701dca2","shared_citers":40},{"title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","work_id":"e6b75ad5-2877-4168-97c8-710407094d20","shared_citers":36},{"title":"Qwen3 Technical Report","work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e","shared_citers":28},{"title":"Chain-of-Thought Prompting Elicits Reasoning in Large Language Models","work_id":"d1cf6693-a082-403c-ada9-dac7b96341f9","shared_citers":27},{"title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","work_id":"c5006563-f3ec-438a-9e35-b7b484f34828","shared_citers":25},{"title":"GPT-4 Technical Report","work_id":"b928e041-6991-4c08-8c81-0359e4097c7b","shared_citers":22},{"title":"Least-to-Most Prompting Enables Complex Reasoning in Large Language Models","work_id":"7e58c111-4666-4996-b5ad-1c8efd433083","shared_citers":22},{"title":"Measuring Mathematical Problem Solving With the MATH Dataset","work_id":"50652ac6-fb7c-4675-a2c2-159c241feb17","shared_citers":22},{"title":"Evaluating Large Language Models Trained on Code","work_id":"042493e9-b26f-4b4e-bbde-382072ca9b08","shared_citers":21},{"title":"ReAct: Synergizing Reasoning and Acting in Language Models","work_id":"407a2351-25f1-497d-b611-f77d0292a8e6","shared_citers":20},{"title":"Scaling LLM Test-Time Compute Optimally can be More Effective than Scaling Model Parameters","work_id":"a8d50b24-bdf5-46ed-bc4f-2927dfd81f1d","shared_citers":19},{"title":"The Llama 3 Herd of Models","work_id":"1549a635-88af-4ac1-acfe-51ae7bb53345","shared_citers":16},{"title":"Self-Refine: Iterative Refinement with Self-Feedback","work_id":"59181e7f-e58e-45d3-8146-4477a9f53d5a","shared_citers":15},{"title":"Language Models (Mostly) Know What They Know","work_id":"8ca58a10-da41-4f70-baae-7e449512e345","shared_citers":14},{"title":"LLaMA: Open and Efficient Foundation Language Models","work_id":"c018fc23-6f3f-4035-9d02-28a2173b2b9d","shared_citers":14},{"title":"Large Language Models are Zero-Shot Reasoners","work_id":"d9b7eb1a-7165-46ff-9f06-d2f0b9d6f95d","shared_citers":13},{"title":"Qwen2.5 Technical Report","work_id":"d8432992-4980-4a81-85c7-9fa2c2b87f85","shared_citers":13},{"title":"Scaling Laws for Neural Language Models","work_id":"b7dd8749-9c45-4977-ab9b-64478dce1ae8","shared_citers":13},{"title":"Training language models to follow instructions with human feedback","work_id":"52aff42f-4fa9-4fcf-bdb3-1459b9bebf65","shared_citers":13},{"title":"DAPO: An Open-Source LLM Reinforcement Learning System at Scale","work_id":"64019d00-0b11-4bbd-b173-b46c8fad0157","shared_citers":12},{"title":"Reflexion: Language Agents with Verbal Reinforcement Learning","work_id":"778f739e-5f55-4961-8a2a-e4736a2757f4","shared_citers":12},{"title":"OpenAI o1 System Card","work_id":"68d3c334-0fc9-49e3-b7b0-a69afae933e2","shared_citers":11},{"title":"Program of Thoughts Prompting: Disentangling Computation from Reasoning for Numerical Reasoning Tasks","work_id":"618aa44c-a6c6-425c-abce-8aa8aa842921","shared_citers":11},{"title":"Proximal Policy Optimization Algorithms","work_id":"240c67fe-d14d-4520-91c1-38a4e272ca19","shared_citers":11}],"time_series":[{"n":4,"year":2022},{"n":5,"year":2023},{"n":6,"year":2024},{"n":1,"year":2025},{"n":136,"year":2026}]},"authors":[{"id":"c5de7f09-6e32-4784-8fa7-9532f012735a","orcid":null,"display_name":"Dale Schuurmans","source":"manual","import_confidence":0.72},{"id":"46959ee3-5c32-49cb-becf-e7baf14d04f2","orcid":null,"display_name":"Ed Chi","source":"manual","import_confidence":0.72},{"id":"cb0687c7-02a4-474c-a3bd-ba5a6e0d00f9","orcid":null,"display_name":"Jason Wei","source":"manual","import_confidence":0.72},{"id":"b7d93d8f-b3b6-4571-b881-819e8f1bae2d","orcid":null,"display_name":"Quoc Le","source":"manual","import_confidence":0.72},{"id":"e0e3c462-ade2-43d1-974e-1a346b19be3b","orcid":null,"display_name":"Sharan Narang","source":"manual","import_confidence":0.72},{"id":"7d44fff2-95d1-41fc-83c5-2969ba35b464","orcid":null,"display_name":"Xuezhi Wang","source":"manual","import_confidence":0.72}]}}