{"id":"8bf8c3d6-0e2e-41e0-b8d7-950fe518bbcd","arxiv_id":"2608.13221","paper_version":1,"verdict":"CONDITIONAL","confidence":"HIGH","novelty_score":6.0,"correctness_risk":"medium","formal_verification":"none","parameter_count":2,"one_line_summary":"A new benchmark parses LLM reasoning traces on Go life-and-death problems into search trees and shows that search organization, not token volume, distinguishes stronger reasoners.","lead":"This paper introduces TsuGO, a benchmark that uses Go life-and-death puzzles to measure how well LLMs organize their search through alternatives. It reports that current models are far from stable solving and that a new Search Efficiency score tracks accuracy better than token cost alone.","discovery_kind":"new_method","skeptic_critique":null,"referee_report":null,"author_rebuttal":null,"desk_editor":null,"rs_alignment":null,"lean_confirmation":null,"pith_extraction":null,"created_at":"2026-08-14T15:59:17.793695+00:00","model_set":{"reader":"deepseek-v4-flash"},"falsifier":null,"supporting_citations":[],"review_version":1}