{"total":14,"items":[{"citing_arxiv_id":"2607.00760","ref_index":34,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"MosaicKV: Serving Long-Context LLM with Dynamic Two-D KV Cache Compression","primary_cat":"cs.LG","submitted_at":"2026-07-01T10:44:57+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"MosaicKV achieves up to 16x attention speedup, 4.8x lower decode latency, 7.3x higher throughput, and 3x memory reduction with 1.76% accuracy loss via dynamic two-D KV cache compression and management on H800 GPUs.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.30571","ref_index":3,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Attractor States Emerge in Multi-Turn LLM Conversations","primary_cat":"cs.LG","submitted_at":"2026-06-29T17:14:29+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Self-play LLM trajectories form model-specific attractors that asymmetrically influence mixed-play partners' stylistic choices and stances across 7 models and 20 topics.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.19659","ref_index":10,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"SAGE-OPD: Selective Agent-Guided Intervention for Multi-Turn On-Policy Distillation","primary_cat":"cs.CL","submitted_at":"2026-06-17T23:58:14+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"SAGE-OPD improves multi-turn OPD via turn-level selective intervention, teacher-confidence weighting, and loss normalization, reporting up to 13.3% relative gain in ALFWorld unseen success rate over standard OPD.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.06302","ref_index":27,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Tangram: Unlocking Non-Uniform KV Cache for Efficient Multi-turn LLM Serving","primary_cat":"cs.LG","submitted_at":"2026-06-04T15:41:27+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"Tangram makes non-uniform KV cache compression practical for LLM serving with deterministic budget allocation, head group paging, and ahead-of-time load balancing, achieving up to 2.6x throughput gains.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.31455","ref_index":12,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"DRIFT: Decoupled Rollouts and Importance-Weighted Fine-Tuning for Efficient Multi-Turn Optimization","primary_cat":"cs.LG","submitted_at":"2026-05-29T15:49:13+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"DRIFT achieves multi-turn RL performance via offline importance-weighted SFT by leveraging the equivalence of KL-regularized RL to weighted supervised learning.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.22612","ref_index":8,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Healthcare LLM Benchmarks Are Only as Good as Their Explicit Assumptions","primary_cat":"cs.CY","submitted_at":"2026-05-21T15:27:58+00:00","verdict":"CONDITIONAL","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Healthcare LLM benchmarks overlook implicit assumptions about user behavior that split into task assumptions testable from conversation data and outcome assumptions requiring behavioral studies, shown by reanalyzing an RCT where both gaps are roughly equal.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.21748","ref_index":15,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"RankJudge: A Multi-Turn LLM-as-a-Judge Synthetic Benchmark Generator","primary_cat":"cs.CL","submitted_at":"2026-05-20T21:20:01+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"RankJudge creates paired multi-turn conversations with isolated single-turn flaws to generate unambiguous benchmarks for LLM-as-a-judge systems across ML, biomedicine, and finance domains.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.17329","ref_index":18,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"LPG: Balancing Efficiency and Policy Reasoning in Latent Policy Guardrails","primary_cat":"cs.CR","submitted_at":"2026-05-17T08:35:38+00:00","verdict":"CONDITIONAL","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"LPG compresses policy deliberation into 10 latent tokens to reach 84.5% safety accuracy and 11x speedup over explicit reasoning baselines on guardrail benchmarks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.11317","ref_index":28,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"SOMA: Efficient Multi-turn LLM Serving via Small Language Model","primary_cat":"cs.CL","submitted_at":"2026-05-11T23:07:33+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"SOMA estimates a local response manifold from early turns and adapts a small surrogate model via divergence-maximizing prompts and localized LoRA fine-tuning for efficient multi-turn serving.","context_count":1,"top_context_role":"background","top_context_polarity":"background","context_text":"[26] Huayang Li, Tian Lan, Zihao Fu, Deng Cai, Lemao Liu, Nigel Collier, Taro Watanabe, and Yixuan Su. Repetition in repetition out: Towards understanding neural text degeneration from the data perspective.Advances in Neural Information Processing Systems, 36:72888-72903, 2023. [27] Lincan Li, Zheng Chen, and Yushun Dong. Llm as clinical graph structure refiner: Enhancing representation learning in eeg seizure diagnosis, 2026. [28] Yubo Li, Xiaobin Shen, Xinyu Yao, Xueying Ding, Yidi Miao, Ramayya Krishnan, and Rema Padman. Beyond single-turn: A survey on multi-turn interactions with large language models. arXiv preprint arXiv:2504.04717, 2025. [29] Fang Liu, Yang Liu, Lin Shi, Houkun Huang, Ruifeng Wang, Zhen Yang, Li Zhang, Zhongqi Li, and Yuchi Ma. Exploring and evaluating hallucinations in llm-powered code generation."},{"citing_arxiv_id":"2605.11002","ref_index":22,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"MT-JailBench: A Modular Benchmark for Understanding Multi-Turn Jailbreak Attacks","primary_cat":"cs.CR","submitted_at":"2026-05-10T00:17:14+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"MT-JailBench is a modular benchmark that standardizes evaluation of multi-turn jailbreaks to identify key success drivers and enable stronger combined attacks.","context_count":1,"top_context_role":"background","top_context_polarity":"background","context_text":"merely longer versions of single-turn jailbreaks, but because they exploit the same mechanism that makes conversational LLMs useful: the ability to accumulate context, infer intent, and act on it. Despite the growing safety concern posed by multi-turn attacks, there has been little systematic study of how existing methods compare when surrounding conditions are held fixed [22]. As a result, it often remains unclear what makes these attacks work and which failure modes defenses should target. Most attacks are introduced as complete pipelines, each with its own budget, judge, retry rule, and strategy generation procedure. One method may be allowed more turns, another may restart from many sampled strategies, and another may use a more permissive judge."},{"citing_arxiv_id":"2605.05413","ref_index":12,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"From History to State: Constant-Context Skill Learning for LLM Agents","primary_cat":"cs.AI","submitted_at":"2026-05-06T20:13:42+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Constant-context skill learning trains reusable task-family modules for LLM agents using a deterministic state block for progress tracking and subgoal rewards, achieving 89.6% unseen success on ALFWorld, 76.8% on WebShop, and 66.4% on SciWorld with Qwen3-8B while reducing prompt tokens 2-7x.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.20996","ref_index":3,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"AFRILANGTUTOR: Advancing Language Tutoring and Culture Education in Low-Resource Languages with Large Language Models","primary_cat":"cs.CL","submitted_at":"2026-04-22T18:38:04+00:00","verdict":"CONDITIONAL","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Fine-tuning Llama-3-8B-IT and Gemma-3-12B-IT on automatically generated tutoring data from African language dictionaries yields 1.8–15.5% improvements over base models under LLM-as-a-judge evaluation across 10 African languages.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.10027","ref_index":8,"ref_count":2,"confidence":0.98,"is_internal_anchor":true,"paper_title":"SinkTrack: Attention Sink based Context Anchoring for Large Language Models","primary_cat":"cs.CV","submitted_at":"2026-04-11T04:49:11+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"SinkTrack anchors LLMs to initial context by modifying the attention sink token with injected features, yielding gains on textual and multimodal tasks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.13061","ref_index":16,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Token Statistics Reveal Conversational Drift in Multi-turn LLM Interaction","primary_cat":"cs.CL","submitted_at":"2026-03-18T18:10:37+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Bipredictability from token statistics monitors structural consistency in multi-turn LLM interactions, showing 85% alignment with structure but only 44% with semantics and 100% sensitivity to tested drifts across 4574 turns.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null}],"limit":50,"offset":0}