{"total":2,"items":[{"citing_arxiv_id":"2607.08340","ref_index":5,"ref_count":1,"confidence":0.88,"is_internal_anchor":false,"paper_title":"Spectral Analysis of Dueling Q-Learning","primary_cat":"cs.LG","submitted_at":"2026-07-09T10:29:22+00:00","verdict":"CONDITIONAL","verdict_confidence":"HIGH","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Unregularized centered dueling Q-learning converges under a joint spectral radius condition, with value and advantage acting as distinct gains on common and differential Q components.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.27112","ref_index":27,"ref_count":1,"confidence":0.88,"is_internal_anchor":false,"paper_title":"Heavy-Ball Q-Learning with Residual Weighting Correction","primary_cat":"cs.LG","submitted_at":"2026-06-25T14:48:58+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Heavy-ball momentum plus a residual-weighting correction yields a JSR-certified faster mean rate than Q-learning when the projected switching family is strictly faster than the constant all-ones mode.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null}],"limit":50,"offset":0}