{"total":15,"items":[{"citing_arxiv_id":"2607.06145","ref_index":18,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Prompting Complexity: Shortest Prompts for Texts and Behaviors in LLMs","primary_cat":"cs.CL","submitted_at":"2026-07-07T11:12:44+00:00","verdict":"CONDITIONAL","verdict_confidence":"HIGH","novelty_score":6.0,"formal_verification":"none","one_line_summary":"The paper defines prompting complexity as the length of the shortest plausible prompt that deterministically generates a target text with a fixed language model.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.00053","ref_index":2,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"SWE-Router: Routing in Multi-turn Agentic Software Engineering Tasks","primary_cat":"cs.SE","submitted_at":"2026-06-30T01:46:26+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"SWE-Router introduces trajectory-conditioned value-based routing for LLM agents on SWE tasks, with a Bayes-optimality theorem and empirical cost savings while retaining most strong-model performance.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.26836","ref_index":6,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"The Capability Frontier: Benchmarks Miss 82% of Model Performance","primary_cat":"cs.AI","submitted_at":"2026-06-25T10:20:47+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"The Capability Frontier shows standard LLM benchmarks underestimate performance by 82% by ignoring model specialization across questions and the benefits of multiple generations per query.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.17949","ref_index":29,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"RouteBalance: Fused Model Routing and Load Balancing for Heterogeneous LLM Serving","primary_cat":"cs.DC","submitted_at":"2026-06-16T14:02:34+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"RouteBalance fuses routing and load balancing for heterogeneous LLM serving and traces the upper quality-cost-throughput frontier on a 13-instance 28-GPU cluster.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.07711","ref_index":8,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Rosetta Memory: Adaptive Memory for Cross-LLM Agents","primary_cat":"cs.LG","submitted_at":"2026-06-05T13:50:59+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"Rosetta Memory trains two profile-conditioned operators with a minimum-gain sampling curriculum and performance-gap reward to enable memory transfer between LLMs, showing gains on multi-hop QA benchmarks and robustness to unseen models.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.06098","ref_index":14,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"IR3DE: A Linear Router for Large Language Models","primary_cat":"cs.CL","submitted_at":"2026-06-04T12:36:28+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"IR3DE is a ridge regression router for domain-expert LLMs that matches or exceeds baselines in language modeling and reasoning tasks while allowing dynamic expert addition or removal without retraining.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.07587","ref_index":25,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"The Routing Plateau: Understanding and Breaking the Accuracy Limits of LLM Routers","primary_cat":"cs.LG","submitted_at":"2026-05-27T19:29:01+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"LLM routers across 21 methods on 5 benchmarks converge to similar accuracy below oracle due to learning global performance trends rather than fine-grained query signals.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.14241","ref_index":10,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Latency-Quality Routing for Functionally Equivalent Tools in LLM Agents","primary_cat":"cs.LG","submitted_at":"2026-05-14T01:14:13+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"LQM-ContextRoute routes LLM tool calls via latency-quality matching in a contextual bandit, improving F1 by 2.18 pp, accuracy by up to 18 pp, and NDCG by 2.91-3.22 pp over SW-UCB on web-search, StrategyQA, and retriever benchmarks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.07805","ref_index":10,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Flexible Routing via Uncertainty Decomposition","primary_cat":"cs.LG","submitted_at":"2026-05-08T14:39:57+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"A router that decomposes uncertainty to flexibly route queries between cheap models and oracles while providing regret bounds and supporting abstention in classification tasks with multiple annotations.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.07180","ref_index":34,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Learning Agent Routing From Early Experience","primary_cat":"cs.CL","submitted_at":"2026-05-08T03:18:40+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"BoundaryRouter routes queries to LLM or agent using early experience memory from a seed set, cutting inference time 60.6% versus always using agents and raising performance 28.6% versus always using direct LLM inference.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.07171","ref_index":16,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Cost-Ordered Feasibility for Multi-Armed Bandits with Cost Subsidy","primary_cat":"cs.LG","submitted_at":"2026-05-08T03:07:25+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Develops COF algorithm for MAB-CS that intelligently checks cheap arm feasibility by pooling samples, with generalized instance-dependent lower bounds and matching upper bounds on cumulative cost and quality regret.","context_count":1,"top_context_role":"background","top_context_polarity":"background","context_text":"the cumulative cost subject to a constraint on the observed reward or quality level. For this reason, model routers are needed to map queries to the most appropriate LLM from a large selection of open source, proprietary, and task-specialized models [21]. The cost of these LLMs may vary over orders of magnitude, therefore routing must balance between expected quality and cost [16, 28, 31, 32]. Cost-subsidy framework:The quality-constrained cost-minimization setting is captured by the recently introduced multi-armed bandits with cost-subsidy (MAB-CS) framework [24, 17]. MAB-CS captures the quality constraint through threshold µCS. For a K-armed bandit instance with arms A={a 1, a2, . . . , aK}, the expected reward from sampling arm k is denoted µk."},{"citing_arxiv_id":"2605.07112","ref_index":16,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Switchcraft: AI Model Router for Agentic Tool Calling","primary_cat":"cs.AI","submitted_at":"2026-05-08T01:41:31+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"Switchcraft routes agentic tool-calling queries to the lowest-cost model that preserves correctness, reaching 82.9% accuracy and 84% cost reduction on five benchmarks.","context_count":1,"top_context_role":"background","top_context_polarity":"background","context_text":"factorization [22, 38], graph neural networks [10], k-means clustering [14], item response theory [30], or constrained optimization [18]. Complementary lines target cost estimation directly [29, 19], enrich the action space with token-budget control [ 34] or best-of-n sampling [8], learn more expressive query representations [33], or improve explainability [21]. Two recent benchmarks also target model routing [16, 17]. None of these works, however, targets tool calling or treats correctness as a first-class objective. As we argue in §2, assuming partial responses are acceptable does not work well in tool calling workloads, where a single incorrect argument can have consequences. We bridge these two threads with Switchcraft-to the best of our knowledge, the first model router"},{"citing_arxiv_id":"2604.15728","ref_index":9,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Privacy-Preserving LLMs Routing","primary_cat":"cs.CR","submitted_at":"2026-04-17T06:02:27+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"PPRoute achieves plaintext-level LLM routing quality with MPC-based privacy and a 20x speedup over naive encrypted implementations via MPC-friendly encoders, multi-step training, and O(1) communication Top-k search.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.15499","ref_index":16,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"SecureRouter: Encrypted Routing for Efficient Secure Inference","primary_cat":"cs.CR","submitted_at":"2026-04-16T20:18:12+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"SecureRouter accelerates secure transformer inference by 1.95x via an encrypted router that selects input-adaptive models from an MPC-optimized pool with negligible accuracy loss.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2507.14200","ref_index":31,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"A Scalable Multi-LLM Collaboration System with Retrieval-based Selection and Exploration-Exploitation-Driven Enhancement","primary_cat":"cs.CL","submitted_at":"2025-07-14T16:17:11+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"SMCS coordinates 15 open-source LLMs via retrieval-based prior selection and exploration-exploitation posterior enhancement, outperforming GPT-4.1 by 5.36% and GPT-o3-mini by 5.28% on eight benchmarks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null}],"limit":50,"offset":0}