{"id":"feab6b6f-0028-496f-bba4-c4d38d968526","arxiv_id":"2504.11997","paper_version":1,"verdict":"CONDITIONAL","confidence":"MODERATE","novelty_score":6.0,"correctness_risk":"medium","formal_verification":"none","parameter_count":0,"one_line_summary":"A discounted value-iteration algorithm with visited-state clipping and deviation-controlled updates achieves ~O(sp(v*) sqrt(d^3 T)) regret for infinite-horizon average-reward linear MDPs with computational cost independent of the state-space size.","lead":"This paper proposes a new reinforcement learning algorithm for long-run average reward in environments with linear structure. It matches the best known regret rate while removing the previous need to compute a value-function minimum over the whole state space, making the computation independent of state-space size.","discovery_kind":"extension","skeptic_critique":null,"referee_report":null,"author_rebuttal":null,"desk_editor":null,"rs_alignment":null,"lean_confirmation":null,"pith_extraction":null,"created_at":"2026-08-16T12:42:03.045170+00:00","model_set":{"reader":"deepseek-v4-flash"},"falsifier":null,"supporting_citations":[],"review_version":1}