{"work":{"id":"acab1aa8-b4d6-40e0-a3ee-25341701dca2","openalex_id":null,"doi":"10.1002/j.1545-","arxiv_id":"2110.14168","raw_key":null,"title":"Training Verifiers to Solve Math Word Problems","authors":null,"authors_text":"Karl Cobbe, Vineet Kosaraju, Mohammad Bavarian, Mark Chen, Heewoo Jun, Lukasz Kaiser, Matthias Plappert, Jerry Tworek, Jacob Hilton, Reiichiro Nakano, Christopher Hesse, and John Schulman","year":2021,"venue":"cs.LG","abstract":"State-of-the-art language models can match human performance on many tasks, but they still struggle to robustly perform multi-step mathematical reasoning. To diagnose the failures of current models and support research, we introduce GSM8K, a dataset of 8.5K high quality linguistically diverse grade school math word problems. We find that even the largest transformer models fail to achieve high test performance, despite the conceptual simplicity of this problem distribution. To increase performance, we propose training verifiers to judge the correctness of model completions. At test time, we generate many candidate solutions and select the one ranked highest by the verifier. We demonstrate that verification significantly improves performance on GSM8K, and we provide strong empirical evidence that verification scales more effectively with increased data than a finetuning baseline.","external_url":"https://arxiv.org/abs/2110.14168","cited_by_count":null,"metadata_source":"pith","metadata_fetched_at":"2026-07-11T03:27:47.058704+00:00","pith_arxiv_id":"2110.14168","created_at":"2026-05-09T03:55:08.450539+00:00","updated_at":"2026-07-11T11:50:26.030339+00:00","title_quality_ok":true,"display_title":"Training Verifiers to Solve Math Word Problems","render_title":"Training Verifiers to Solve Math Word Problems"},"hub":{"state":{"work_id":"acab1aa8-b4d6-40e0-a3ee-25341701dca2","tier":"mega_hub","tier_reason":"1,000+ Pith inbound or 100,000+ external citations","pith_inbound_count":1496,"external_cited_by_count":null,"distinct_field_count":34,"first_pith_cited_at":"2021-12-01T22:24:34+00:00","last_pith_cited_at":"2026-07-09T17:35:02+00:00","author_build_status":"needed","summary_status":"needed","contexts_status":"needed","graph_status":"needed","ask_index_status":"needed","reader_status":"needed","recognition_status":"needed","updated_at":"2026-08-21T01:39:39.240895+00:00","tier_text":"mega_hub"},"tier":"mega_hub","role_counts":[{"context_role":"background","n":126},{"context_role":"dataset","n":101},{"context_role":"method","n":7},{"context_role":"baseline","n":4},{"context_role":"other","n":2}],"polarity_counts":[{"context_polarity":"background","n":114},{"context_polarity":"use_dataset","n":99},{"context_polarity":"unclear","n":16},{"context_polarity":"use_method","n":7},{"context_polarity":"baseline","n":4}],"runs":{"ask_index":{"job_type":"ask_index","status":"succeeded","result":{"title":"Training Verifiers to Solve Math Word Problems","claims":[{"claim_text":"State-of-the-art language models can match human performance on many tasks, but they still struggle to robustly perform multi-step mathematical reasoning. To diagnose the failures of current models and support research, we introduce GSM8K, a dataset of 8.5K high quality linguistically diverse grade school math word problems. We find that even the largest transformer models fail to achieve high test performance, despite the conceptual simplicity of this problem distribution. To increase performance, we propose training verifiers to judge the correctness of model completions. At test time, we ge","claim_type":"abstract","evidence_strength":"source_metadata"}],"why_cited":"Pith tracks Training Verifiers to Solve Math Word Problems because it crossed a citation-hub threshold.","role_counts":[]},"error":null,"updated_at":"2026-05-13T18:03:31.006350+00:00"},"author_expand":{"job_type":"author_expand","status":"succeeded","result":{"authors_linked":[{"id":"07c47add-2301-4164-9d06-23347fc20617","orcid":null,"display_name":"Karl Cobbe"},{"id":"edf3b705-ff6d-4713-9b26-27729234c00d","orcid":null,"display_name":"Vineet Kosaraju"},{"id":"9253b15a-b5df-4d8c-bad6-79bec0dcd54d","orcid":null,"display_name":"Mohammad Bavarian"},{"id":"27b716ab-b5bb-4619-9617-be39d50e5f88","orcid":null,"display_name":"Mark Chen"},{"id":"dfca2058-03d1-4251-92fe-0eaddf1dfcf0","orcid":null,"display_name":"Heewoo Jun"},{"id":"7b67bce8-4222-4c96-93ed-a2ccbbe6513d","orcid":null,"display_name":"Lukasz Kaiser"}]},"error":null,"updated_at":"2026-05-13T17:24:05.834629+00:00"},"context_extract":{"job_type":"context_extract","status":"succeeded","result":{"enqueued_papers":25},"error":null,"updated_at":"2026-05-13T17:53:40.535835+00:00"},"graph_features":{"job_type":"graph_features","status":"succeeded","result":{"co_cited":[{"title":"Evaluating Large Language Models Trained on Code","work_id":"042493e9-b26f-4b4e-bbde-382072ca9b08","shared_citers":139},{"title":"Qwen3 Technical Report","work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e","shared_citers":115},{"title":"The Llama 3 Herd of Models","work_id":"1549a635-88af-4ac1-acfe-51ae7bb53345","shared_citers":113},{"title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","work_id":"c5006563-f3ec-438a-9e35-b7b484f34828","shared_citers":107},{"title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","work_id":"e6b75ad5-2877-4168-97c8-710407094d20","shared_citers":104},{"title":"GPT-4 Technical Report","work_id":"b928e041-6991-4c08-8c81-0359e4097c7b","shared_citers":78},{"title":"Measuring Mathematical Problem Solving With the MATH Dataset","work_id":"50652ac6-fb7c-4675-a2c2-159c241feb17","shared_citers":77},{"title":"Think you have Solved Question Answering? Try ARC, the AI2 Reasoning Challenge","work_id":"28ea1282-d657-4c61-a83c-f1249be6d6b1","shared_citers":77},{"title":"Program Synthesis with Large Language Models","work_id":"fd241a05-03b9-4de2-9588-9d77ce176125","shared_citers":70},{"title":"Proximal Policy Optimization Algorithms","work_id":"240c67fe-d14d-4520-91c1-38a4e272ca19","shared_citers":65},{"title":"Measuring Massive Multitask Language Understanding","work_id":"e87ec49a-544b-4ec8-8991-75298c64ff5e","shared_citers":61},{"title":"Qwen2.5 Technical Report","work_id":"d8432992-4980-4a81-85c7-9fa2c2b87f85","shared_citers":57},{"title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","work_id":"68a5177f-d644-44c1-bd4f-4e5278c22f5d","shared_citers":54},{"title":"DeepSeek-V3 Technical Report","work_id":"57d2791d-2219-4c31-a077-afc04b12a75c","shared_citers":48},{"title":"LLaMA: Open and Efficient Foundation Language Models","work_id":"c018fc23-6f3f-4035-9d02-28a2173b2b9d","shared_citers":48},{"title":"Scaling Laws for Neural Language Models","work_id":"b7dd8749-9c45-4977-ab9b-64478dce1ae8","shared_citers":47},{"title":"Mistral 7B","work_id":"eb5e1305-ad11-4875-ad8d-ad8b8f697599","shared_citers":40},{"title":"Self-Consistency Improves Chain of Thought Reasoning in Language Models","work_id":"8c6d5a6b-b5cc-4105-9c84-9c34bb9375bb","shared_citers":38},{"title":"Challenging BIG-Bench Tasks and Whether Chain-of-Thought Can Solve Them","work_id":"513eb205-04ca-4722-9a43-a74e8cbe7e85","shared_citers":35},{"title":"DAPO: An Open-Source LLM Reinforcement Learning System at Scale","work_id":"64019d00-0b11-4bbd-b173-b46c8fad0157","shared_citers":35},{"title":"Scaling LLM Test-Time Compute Optimally can be More Effective than Scaling Model Parameters","work_id":"a8d50b24-bdf5-46ed-bc4f-2927dfd81f1d","shared_citers":35},{"title":"Instruction-Following Evaluation for Large Language Models","work_id":"3aa06177-125a-4f5a-8f4a-8070c5986c26","shared_citers":33},{"title":"Training a Helpful and Harmless Assistant with Reinforcement Learning from Human Feedback","work_id":"a1f2574b-a899-4713-be60-c87ba332656c","shared_citers":33},{"title":"Let's Verify Step by Step","work_id":"6d05b790-04c5-4fd2-91b2-ba1dfdd5770f","shared_citers":32}],"time_series":[{"n":1,"year":2021},{"n":6,"year":2022},{"n":16,"year":2023},{"n":30,"year":2024},{"n":16,"year":2025},{"n":401,"year":2026}]},"error":null,"updated_at":"2026-05-13T17:25:55.152933+00:00"},"identity_refresh":{"job_type":"identity_refresh","status":"succeeded","result":{"fixed":1,"items":[{"title":"Qwen3 Technical Report","work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e","resolver":"local_arxiv","confidence":0.98,"old_work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e"}],"errors":[],"attempted":1},"error":null,"updated_at":"2026-05-13T17:53:39.755816+00:00"},"reader_index":{"job_type":"reader_index","status":"succeeded","result":{"note":"annotated reader requires full-text/OA fetch; shell is wired for mega hubs","status":"reader queued"},"error":null,"updated_at":"2026-06-29T00:28:12.424078+00:00"},"recognition_alignment":{"job_type":"recognition_alignment","status":"succeeded","result":{"modules":["IndisputableMonolith.Sport.PeakPerformanceFromJCost","IndisputableMonolith.Sports.PeakPerformanceFromPhiLadder","IndisputableMonolith.Cognition.AnimalZComplexityBound","IndisputableMonolith.Information.ChurchTuring","IndisputableMonolith.Education.MasteryThresholdFromGap45","IndisputableMonolith.Flight.Falsifiers","IndisputableMonolith.Materials.RoomTSuperconductorCandidate","IndisputableMonolith.MusicTheory.Rhythm"],"query_chars":938},"error":null,"updated_at":"2026-06-29T00:28:34.193407+00:00"},"role_polarity":{"job_type":"role_polarity","status":"succeeded","result":{"title":"Training Verifiers to Solve Math Word Problems","claims":[{"claim_text":"State-of-the-art language models can match human performance on many tasks, but they still struggle to robustly perform multi-step mathematical reasoning. To diagnose the failures of current models and support research, we introduce GSM8K, a dataset of 8.5K high quality linguistically diverse grade school math word problems. We find that even the largest transformer models fail to achieve high test performance, despite the conceptual simplicity of this problem distribution. To increase performance, we propose training verifiers to judge the correctness of model completions. At test time, we ge","claim_type":"abstract","evidence_strength":"source_metadata"}],"why_cited":"Pith tracks Training Verifiers to Solve Math Word Problems because it crossed a citation-hub threshold.","role_counts":[]},"error":null,"updated_at":"2026-05-13T18:03:31.004378+00:00"},"summary_claims":{"job_type":"summary_claims","status":"succeeded","result":{"title":"Training Verifiers to Solve Math Word Problems","claims":[{"claim_text":"State-of-the-art language models can match human performance on many tasks, but they still struggle to robustly perform multi-step mathematical reasoning. To diagnose the failures of current models and support research, we introduce GSM8K, a dataset of 8.5K high quality linguistically diverse grade school math word problems. We find that even the largest transformer models fail to achieve high test performance, despite the conceptual simplicity of this problem distribution. To increase performance, we propose training verifiers to judge the correctness of model completions. At test time, we ge","claim_type":"abstract","evidence_strength":"source_metadata"}],"why_cited":"Pith tracks Training Verifiers to Solve Math Word Problems because it crossed a citation-hub threshold.","role_counts":[]},"error":null,"updated_at":"2026-05-13T17:25:52.713246+00:00"}},"summary":{"title":"Training Verifiers to Solve Math Word Problems","claims":[{"claim_text":"State-of-the-art language models can match human performance on many tasks, but they still struggle to robustly perform multi-step mathematical reasoning. To diagnose the failures of current models and support research, we introduce GSM8K, a dataset of 8.5K high quality linguistically diverse grade school math word problems. We find that even the largest transformer models fail to achieve high test performance, despite the conceptual simplicity of this problem distribution. To increase performance, we propose training verifiers to judge the correctness of model completions. At test time, we ge","claim_type":"abstract","evidence_strength":"source_metadata"}],"why_cited":"Pith tracks Training Verifiers to Solve Math Word Problems because it crossed a citation-hub threshold.","role_counts":[]},"graph":{"co_cited":[{"title":"Evaluating Large Language Models Trained on Code","work_id":"042493e9-b26f-4b4e-bbde-382072ca9b08","shared_citers":139},{"title":"Qwen3 Technical Report","work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e","shared_citers":115},{"title":"The Llama 3 Herd of Models","work_id":"1549a635-88af-4ac1-acfe-51ae7bb53345","shared_citers":113},{"title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","work_id":"c5006563-f3ec-438a-9e35-b7b484f34828","shared_citers":107},{"title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","work_id":"e6b75ad5-2877-4168-97c8-710407094d20","shared_citers":104},{"title":"GPT-4 Technical Report","work_id":"b928e041-6991-4c08-8c81-0359e4097c7b","shared_citers":78},{"title":"Measuring Mathematical Problem Solving With the MATH Dataset","work_id":"50652ac6-fb7c-4675-a2c2-159c241feb17","shared_citers":77},{"title":"Think you have Solved Question Answering? Try ARC, the AI2 Reasoning Challenge","work_id":"28ea1282-d657-4c61-a83c-f1249be6d6b1","shared_citers":77},{"title":"Program Synthesis with Large Language Models","work_id":"fd241a05-03b9-4de2-9588-9d77ce176125","shared_citers":70},{"title":"Proximal Policy Optimization Algorithms","work_id":"240c67fe-d14d-4520-91c1-38a4e272ca19","shared_citers":65},{"title":"Measuring Massive Multitask Language Understanding","work_id":"e87ec49a-544b-4ec8-8991-75298c64ff5e","shared_citers":61},{"title":"Qwen2.5 Technical Report","work_id":"d8432992-4980-4a81-85c7-9fa2c2b87f85","shared_citers":57},{"title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","work_id":"68a5177f-d644-44c1-bd4f-4e5278c22f5d","shared_citers":54},{"title":"DeepSeek-V3 Technical Report","work_id":"57d2791d-2219-4c31-a077-afc04b12a75c","shared_citers":48},{"title":"LLaMA: Open and Efficient Foundation Language Models","work_id":"c018fc23-6f3f-4035-9d02-28a2173b2b9d","shared_citers":48},{"title":"Scaling Laws for Neural Language Models","work_id":"b7dd8749-9c45-4977-ab9b-64478dce1ae8","shared_citers":47},{"title":"Mistral 7B","work_id":"eb5e1305-ad11-4875-ad8d-ad8b8f697599","shared_citers":40},{"title":"Self-Consistency Improves Chain of Thought Reasoning in Language Models","work_id":"8c6d5a6b-b5cc-4105-9c84-9c34bb9375bb","shared_citers":38},{"title":"Challenging BIG-Bench Tasks and Whether Chain-of-Thought Can Solve Them","work_id":"513eb205-04ca-4722-9a43-a74e8cbe7e85","shared_citers":35},{"title":"DAPO: An Open-Source LLM Reinforcement Learning System at Scale","work_id":"64019d00-0b11-4bbd-b173-b46c8fad0157","shared_citers":35},{"title":"Scaling LLM Test-Time Compute Optimally can be More Effective than Scaling Model Parameters","work_id":"a8d50b24-bdf5-46ed-bc4f-2927dfd81f1d","shared_citers":35},{"title":"Instruction-Following Evaluation for Large Language Models","work_id":"3aa06177-125a-4f5a-8f4a-8070c5986c26","shared_citers":33},{"title":"Training a Helpful and Harmless Assistant with Reinforcement Learning from Human Feedback","work_id":"a1f2574b-a899-4713-be60-c87ba332656c","shared_citers":33},{"title":"Let's Verify Step by Step","work_id":"6d05b790-04c5-4fd2-91b2-ba1dfdd5770f","shared_citers":32}],"time_series":[{"n":1,"year":2021},{"n":6,"year":2022},{"n":16,"year":2023},{"n":30,"year":2024},{"n":16,"year":2025},{"n":401,"year":2026}]},"authors":[{"id":"dfca2058-03d1-4251-92fe-0eaddf1dfcf0","orcid":null,"display_name":"Heewoo Jun","source":"manual","import_confidence":0.72},{"id":"07c47add-2301-4164-9d06-23347fc20617","orcid":null,"display_name":"Karl Cobbe","source":"manual","import_confidence":0.72},{"id":"7b67bce8-4222-4c96-93ed-a2ccbbe6513d","orcid":null,"display_name":"Lukasz Kaiser","source":"manual","import_confidence":0.72},{"id":"27b716ab-b5bb-4619-9617-be39d50e5f88","orcid":null,"display_name":"Mark Chen","source":"manual","import_confidence":0.72},{"id":"9253b15a-b5df-4d8c-bad6-79bec0dcd54d","orcid":null,"display_name":"Mohammad Bavarian","source":"manual","import_confidence":0.72},{"id":"edf3b705-ff6d-4713-9b26-27729234c00d","orcid":null,"display_name":"Vineet Kosaraju","source":"manual","import_confidence":0.72}]},"citers":{"total":1496,"items":[{"citing_arxiv_id":"2607.08734","ref_index":6,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"The Illusion of Equivalency: Statistical Characterization of Quantization Effects in LLMs","primary_cat":"cs.AI","submitted_at":"2026-07-09T17:35:02+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Quantized LLMs diverge from their base models at the decision level even when accuracy is preserved, with query and key attention projections showing the greatest structural distortion under low-bit compression.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.08170","ref_index":7,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Understanding Layer Patching in Model Size Interpolation","primary_cat":"cs.LG","submitted_at":"2026-07-09T07:14:12+00:00","verdict":"CONDITIONAL","verdict_confidence":"HIGH","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Optimal layer-patching order for boomerang distillation is a shortest path on a KL-weighted Boolean lattice; greedy KLPatch and simple sequential orders often yield near-optimal interpolations.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.08116","ref_index":43,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"MORES: Mobile Reasoning-as-a-Service via Distributed LLM Inference-Time Scaling","primary_cat":"cs.NI","submitted_at":"2026-07-09T05:24:31+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":5.5,"formal_verification":"none","one_line_summary":"A device–server split of recurrent latent LLM reasoning plus semantic MoE-SAC scheduling yields about 18% higher simulated system throughput than plain SAC under energy, recurrence, and latency budgets.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.08080","ref_index":22,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"MASTE: A Multi-Agent Pipeline for Zero-Shot Aspect Sentiment Triplet Extraction","primary_cat":"cs.CL","submitted_at":"2026-07-09T03:20:54+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":4.0,"formal_verification":"none","one_line_summary":"A four-agent LLM pipeline decomposes aspect-sentiment triplet extraction into sequential subtasks, outperforming zero-shot baselines on four benchmarks without labeled training data.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.08009","ref_index":34,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"From Execution to Education: A Bloom-Aligned Framework for Measuring Educational Control in LLMs","primary_cat":"cs.CL","submitted_at":"2026-07-09T00:27:07+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":6.5,"formal_verification":"none","one_line_summary":"On 2,520 programming tasks, matched Qwen general and coder models reliably raise Bloom cognitive demand but fail to lower it, so execution skill does not imply educational control.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.07918","ref_index":8,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Efficient Safety Alignment of Language Models via Latent Personality Traits","primary_cat":"cs.LG","submitted_at":"2026-07-08T21:03:27+00:00","verdict":null,"verdict_confidence":null,"novelty_score":null,"formal_verification":null,"one_line_summary":null,"context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.07916","ref_index":15,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Persona Cartography: Charting Language Model Personality Traits in Weight Space","primary_cat":"cs.AI","submitted_at":"2026-07-08T21:00:44+00:00","verdict":"CONDITIONAL","verdict_confidence":"HIGH","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Composable LoRA adapters can amplify or suppress OCEAN traits in LLMs, combine approximately additively, preserve moderate-scale capability, and move safety-relevant behaviours.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.07690","ref_index":14,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Agon: Competitive Cross-Model RL with Implicit Rival Grading of Reasoning","primary_cat":"cs.LG","submitted_at":"2026-07-08T17:49:14+00:00","verdict":"CONDITIONAL","verdict_confidence":"UNKNOWN","novelty_score":7.0,"formal_verification":"none","one_line_summary":"Training two LoRA adapters competitively against each other, where each reads the other's solution summary and is rewarded for out-solving it, doubles GRPO's pass@1 on hard math while shortening reasoning traces.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.07678","ref_index":8,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"How Data Shapes RoPE Frequency Usage: From Positional Scale Matching to Length Generalization","primary_cat":"cs.LG","submitted_at":"2026-07-08T17:38:14+00:00","verdict":"CONDITIONAL","verdict_confidence":"HIGH","novelty_score":7.0,"formal_verification":"none","one_line_summary":"RoPE frequency usage is determined by a data-induced dependency width W, with the optimal frequency scaling as π/W, explaining both learned spectra and the success of position interpolation.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.07674","ref_index":21,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Max Out GRPO Signal: Adaptive Trace Prefix Control for Hard Reasoning Problems","primary_cat":"cs.LG","submitted_at":"2026-07-08T17:32:58+00:00","verdict":"CONDITIONAL","verdict_confidence":"UNKNOWN","novelty_score":6.0,"formal_verification":"none","one_line_summary":"AdaPrefix-GRPO treats solution-prefix length as a feedback controller targeting 50% rollout success rate during GRPO training, then anneals to zero prefix, yielding 1.6–2.1× accuracy gains over vanilla GRPO at matched compute on hard math.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.07646","ref_index":3,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"RL Post-Training Builds Compositional Reasoning Strategies","primary_cat":"cs.AI","submitted_at":"2026-07-08T17:04:42+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":7.0,"formal_verification":"none","one_line_summary":"RL post-training composes primitive rewrite skills into reusable macro and parallel contraction strategies that solve problems inaccessible to the base model under large sampling budgets.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.07508","ref_index":3,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Single-Rollout Asynchronous Optimization for Agentic Reinforcement Learning","primary_cat":"cs.LG","submitted_at":"2026-07-08T15:02:19+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"SAO stabilizes asynchronous RL for LLMs by replacing group-wise sampling with single-rollout updates, token-level importance sampling, and targeted value-model training, outperforming GRPO on reasoning and coding benchmarks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.07409","ref_index":12,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"DeLS-Spec: Decoupled Long-Short Contexts for Parallel Speculative Drafting","primary_cat":"cs.CL","submitted_at":"2026-07-08T13:41:52+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":5.0,"formal_verification":"none","one_line_summary":"DeLS-Spec improves block-parallel speculative decoding by fusing DFlash logits with an independently trained lightweight local head and a unigram prior correction.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.07391","ref_index":14,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"MIRA-Math: A Benchmark for Minimal Information Requesting and Mathematical Reasoning","primary_cat":"cs.AI","submitted_at":"2026-07-08T13:23:56+00:00","verdict":"CONDITIONAL","verdict_confidence":"HIGH","novelty_score":6.0,"formal_verification":"none","one_line_summary":"MIRA-Math introduces a 2,310-instance benchmark isolating the ability of LLMs to request a single missing atomic fact needed to solve an underdetermined mathematical problem and then integrate it into an exact answer.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.07386","ref_index":87,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Sparse Delta Memory: Scaling the State of Linear RNNs through Sparsity","primary_cat":"cs.LG","submitted_at":"2026-07-08T13:17:19+00:00","verdict":"CONDITIONAL","verdict_confidence":"HIGH","novelty_score":6.0,"formal_verification":"none","one_line_summary":"SDM sparsifies the Gated DeltaNet update rule to enable 1000x larger recurrent memory states at iso-FLOP, improving long-context recall and short-context reasoning over GDN and matching full attention at 8B scale.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.07178","ref_index":3,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Entropy Pacing Policy Optimization for Multi-Task Agentic Reinforcement Learning","primary_cat":"cs.LG","submitted_at":"2026-07-08T09:13:05+00:00","verdict":"ACCEPT","verdict_confidence":"HIGH","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Replacing GRPO's fixed clipping range with a task-wise entropy-aware adaptive bound stabilizes multi-task agentic LLM training by synchronizing exploration-exploitation paces.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.07046","ref_index":14,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Voltron: Enabling Elastic Multi-Device Execution of LLM Inference for Empowered Edge Intelligence","primary_cat":"cs.DC","submitted_at":"2026-07-08T06:23:04+00:00","verdict":"CONDITIONAL","verdict_confidence":"UNKNOWN","novelty_score":6.0,"formal_verification":"none","one_line_summary":"A framework called Voltron elastically distributes LLM inference across heterogeneous edge devices using layer-wise hybrid parallelism and mixed precision, achieving up to 16.5% higher accuracy than single-device execution while meeting QoS latency constraints.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.06807","ref_index":21,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"When Agents Go Rogue: Activation-Based Detection of Malicious Behaviors in Multi-Agent Systems","primary_cat":"cs.CR","submitted_at":"2026-07-07T21:12:27+00:00","verdict":"CONDITIONAL","verdict_confidence":"HIGH","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Activation-space divergence detects and corrects compromised LLM agents in multi-agent systems without interaction graphs or synchronized rounds, outperforming graph baselines especially under async stealthy attacks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.06763","ref_index":46,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Trees from Marginals: Autoregressive drafting with factorized priors","primary_cat":"cs.LG","submitted_at":"2026-07-07T19:48:36+00:00","verdict":"ACCEPT","verdict_confidence":"HIGH","novelty_score":7.0,"formal_verification":"none","one_line_summary":"Weaver builds conditional proposal trees from a factorized drafter’s top-K marginals and a rollback-free GDN tree-verify kernel, yielding 4.37× speedup over AR decoding.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.06565","ref_index":58,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"ELSA3D: Elastic Semantic Anchoring for Unified 3D Understanding and Generation","primary_cat":"cs.CV","submitted_at":"2026-07-07T17:59:50+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"ELSA3D introduces elastic semantic anchoring via sparse anchor tokens and a scale-aware octree tokenizer to unify 3D generation and captioning at reduced computational cost.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.06648","ref_index":12,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Final Checkpoints Are Not Enough: Analyzing Latent Reasoning Faithfulness Along Training Trajectories","primary_cat":"cs.LG","submitted_at":"2026-07-07T16:09:44+00:00","verdict":"CONDITIONAL","verdict_confidence":"HIGH","novelty_score":6.5,"formal_verification":"none","one_line_summary":"Latent reasoning faithfulness is a property of training stage and answer format, not of architecture or the final checkpoint alone.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.06088","ref_index":4,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Flow Matching-Based Speech Source Separation with Best-of-N Biometric Sampling","primary_cat":"cs.SD","submitted_at":"2026-07-07T10:00:23+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":5.0,"formal_verification":"none","one_line_summary":"A flow-matching speech separator with biometric best-of-N candidate selection and chunk-wise channel alignment achieves competitive separation metrics and the best downstream ASR/SV error rates among evaluated systems on Libri2Mix.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.05992","ref_index":8,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"PluraMath: Extending Mathematical Reasoning Evaluation Beyond High-Resource Languages","primary_cat":"cs.CL","submitted_at":"2026-07-07T08:25:29+00:00","verdict":"CONDITIONAL","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"PluraMath extends PolyMath with human-validated math problems in 18 mid-to-extreme low-resource languages and benchmarks 27 reasoning LLMs, finding a persistent high- vs low-resource performance gap.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.05863","ref_index":23,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Strategic Bargaining in Multi-Buyer Markets: Reinforcement Learning from Verifiable Rewards for LLM Negotiations","primary_cat":"cs.LG","submitted_at":"2026-07-07T05:41:54+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"RLVR training teaches a 30B LLM to strategically explore a multi-buyer market and extract 70% of available surplus, outperforming frontier models up to 1T parameters in concurrent negotiation.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.06601","ref_index":10,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"TriRoute: Unified Learned Routing for Joint Adaptive Attention, Experts, and KV-Cache Allocation","primary_cat":"cs.LG","submitted_at":"2026-07-07T00:12:46+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"A shared per-token controller jointly routes attention resolution, FFN experts, and KV bit-width and is claimed to Pareto-dominate independently tuned MoD+MoE+KV-quant at matched cost while protecting rare-token accuracy.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.05708","ref_index":12,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Akashic: A Low-Overhead LLM Inference Service with MemAttention","primary_cat":"cs.AI","submitted_at":"2026-07-07T00:06:22+00:00","verdict":"CONDITIONAL","verdict_confidence":"HIGH","novelty_score":5.5,"formal_verification":"none","one_line_summary":"Akashic’s MemAttention plus locality-aware placement improves agent task accuracy by up to 10.2 points and throughput by up to 1.21× over prior memory systems across four long-horizon workloads.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.05391","ref_index":7,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"LLM-as-a-Verifier: A General-Purpose Verification Framework","primary_cat":"cs.AI","submitted_at":"2026-07-06T17:59:35+00:00","verdict":"CONDITIONAL","verdict_confidence":"HIGH","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Expecting over scoring-token logits yields continuous, scalable verification that improves agent trajectory selection and dense RL rewards across coding, robotics, and medical benchmarks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.05355","ref_index":10,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Faithfulness to Refusal: A Causal Audit of Neuron Selectors","primary_cat":"cs.CL","submitted_at":"2026-07-06T17:33:36+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"A causal audit via neuron-row zeroing shows attribution methods (LRP, IG) faithfully identify dispensable neurons and can install refusal behavior, while rank-stability proxies systematically miss selector failures.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.05199","ref_index":2,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Reason, Reward, Refine: Step-Level Errors Corrections with Structured Feedback for Physics Reasoning in Small Language Models","primary_cat":"cs.AI","submitted_at":"2026-07-06T15:16:10+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":5.0,"formal_verification":"none","one_line_summary":"A step-level reward framework using GPT-4o as a training-time verifier reduces reasoning errors in small language models on physics benchmarks by 10-20% over baselines.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.05196","ref_index":128,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Unified Audio Intelligence Without Regressing on Text Intelligence","primary_cat":"cs.CL","submitted_at":"2026-07-06T15:11:57+00:00","verdict":"CONDITIONAL","verdict_confidence":"HIGH","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Audex unifies audio understanding and generation on a strong text MoE backbone with multi-stage SFT plus text-only Cascade RL, matching open SOTA audio scores while mostly retaining text capability.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.02464","ref_index":5,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Will Scaling Improve Social Simulation with LLMs?","primary_cat":"cs.CL","submitted_at":"2026-07-02T17:30:38+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Using 85 controlled and 35 public LLMs, the authors show social-simulation accuracy generally improves with compute, but some behavioral and low-resource tasks do not scale.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.01792","ref_index":19,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"PARTREP: Learning What to Repeat for Decoder-only LLMs","primary_cat":"cs.CL","submitted_at":"2026-07-02T07:07:28+00:00","verdict":"CONDITIONAL","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"PartRep selects high-NLL tokens via a lightweight early-exit gate for partial prompt repetition, retaining most full-repetition gains at 59.4% KV cache and 79% prefill FLOPs on eight benchmarks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.01789","ref_index":7,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"EPnG: Adaptive Expert Prune-and-Grow for Parameter-Efficient MoE Fine-tuning","primary_cat":"cs.LG","submitted_at":"2026-07-02T07:02:44+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"EPnG reallocates LoRA capacity in MoE models by pruning experts with low router gate probabilities and expanding high-importance ones via rank growth, outperforming standard LoRA and nearing full fine-tuning performance with 0.55-0.72% parameters updated.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.01775","ref_index":118,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Set Diffusion: Interpolating Token Orderings Between Autoregression and Diffusion for Fast and Flexible Decoding","primary_cat":"cs.LG","submitted_at":"2026-07-02T06:45:43+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"Set diffusion factorizes likelihood over arbitrary token sets and uses a set-causal diffusion architecture to support KV caching and any-order decoding, yielding improved speed-quality tradeoffs versus prior diffusion LMs.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.01767","ref_index":37,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Repair the Amplifier, Not the Symptom: Stable World-Model Correction for Agent Rollouts","primary_cat":"cs.AI","submitted_at":"2026-07-02T06:31:45+00:00","verdict":"CONDITIONAL","verdict_confidence":"HIGH","novelty_score":6.0,"formal_verification":"none","one_line_summary":"WM-SAR greedily grows a compact connected repair region by marginal residual-spectral relief so that fixing it stabilizes subsequent agent rollouts better than symptom scans under tight token budgets.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.01710","ref_index":17,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Generic Expert Coverage for Pruning SparseMixture-of-Experts Language Models","primary_cat":"cs.AI","submitted_at":"2026-07-02T05:02:18+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Coverage-aware pruning using per-corpus utility profiles on WikiText2 and C4 improves zero-shot accuracy and reduces perplexity degradation in two MoE models at 25-75% retention compared to baselines, without downstream data.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.01612","ref_index":51,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Scaling with Confidence: Calibrating Confidence of LLMs for Adaptive Test Time Scaling","primary_cat":"cs.AI","submitted_at":"2026-07-02T02:29:33+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"C3RL is a new RL algorithm combining correctness, calibration, and reference accuracy rewards to improve LLM confidence calibration, enabling CAS to outperform majority voting with up to 12.33x lower inference cost.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.01511","ref_index":6,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Revisiting Chain-of-Thought Reasoning under Limited Supervision: Semi-supervised Chain-of-Thought Learning","primary_cat":"cs.AI","submitted_at":"2026-07-01T22:17:39+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":4.0,"formal_verification":"none","one_line_summary":"Semi-CoT selects low-entropy pseudo-CoT chains from unlabeled questions via answer-level semantic entropy and shows high pseudo-answer precision but only small or negative gains on math reasoning benchmarks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.01465","ref_index":1,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Beyond Next-Token Prediction: An RLVR Proof of Concept for Tool-Use Agents on Atlassian Workflows","primary_cat":"cs.AI","submitted_at":"2026-07-01T20:55:07+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":4.0,"formal_verification":"none","one_line_summary":"RLVR training on five synthetic Atlassian API environments raises average tool-use reward for Qwen models from 0.35-0.92 to 0.95-1.00 on four non-degenerate scenarios.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.01444","ref_index":4,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"On the Utility and Factual Reliability of Pruned Mixture-of-Experts Models in the Biomedical Domain","primary_cat":"cs.LG","submitted_at":"2026-07-01T20:08:19+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":4.0,"formal_verification":"none","one_line_summary":"Moderate pruning of MoE models preserves in-domain biomedical utility and reliability but both degrade rapidly in cross-domain settings and at extreme pruning ratios.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.01065","ref_index":32,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"GSRQ: Gain-Shape Residual Quantization for Sub-1-bit KV Cache","primary_cat":"cs.LG","submitted_at":"2026-07-01T15:25:21+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"GSRQ applies a gain-shape variant of K-means inside residual quantization to improve directional fidelity, raising LongBench accuracy from 11.34 to 33.54 at 1-bit on LLaMA-3-8B.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.00908","ref_index":3,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Beyond Activation Alignment:The Alignment-Diversity Tradeoff in Task-Aware LLM Quantization","primary_cat":"cs.LG","submitted_at":"2026-07-01T13:12:21+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":7.0,"formal_verification":"none","one_line_summary":"TASA improves task-aware mixed-precision LLM quantization by searching calibration data mixtures via gradient-trace alignment and aggregating perplexity plus reasoning sensitivity signals, enabling 3.5-bit models to match or beat 4-bit baselines with over 20-point gains on GSM8K.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.00664","ref_index":13,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"YOMI-Bench: A Benchmark for Evaluating Kanji Reading and Phonological Understanding of LLMs for Japanese","primary_cat":"cs.CL","submitted_at":"2026-07-01T09:13:19+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"YOMI-Bench is a new benchmark of four tasks for kanji reading and phonological understanding in LLMs, showing low performance even for Japanese-specific and commercial models.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.00572","ref_index":50,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"HARC: Coupling Harmfulness and Refusal Directions for Robust Safety Alignment","primary_cat":"cs.AI","submitted_at":"2026-07-01T07:58:16+00:00","verdict":"ACCEPT","verdict_confidence":"HIGH","novelty_score":7.0,"formal_verification":"none","one_line_summary":"HARC couples harmfulness and refusal directions at prompt and response positions, yielding the best robustness-capability-usability trade-off among major safety methods.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.00531","ref_index":11,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Active-GRPO: Adaptive Imitation and Self-Improving Reasoning for Molecular Optimization","primary_cat":"cs.LG","submitted_at":"2026-07-01T07:22:46+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Active-GRPO reaches 0.1773 average SRxSim on TOMG-Bench MOLOPT by adaptively switching between imitation and self-reinforcement while upgrading references, outperforming GRPO and RePO.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.00466","ref_index":5,"ref_count":2,"confidence":0.98,"is_internal_anchor":true,"paper_title":"ELDR: Expert-Locality-Aware Decode Routing for PD-Disaggregated MoE Serving","primary_cat":"cs.DC","submitted_at":"2026-07-01T05:34:38+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"ELDR reduces median TPOT by 5.9-13.9% in PD-disaggregated MoE serving via expert signatures from prefill, K-means partitioning, and locality-band routing with KV-co-indexed signature cache.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.00276","ref_index":9,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Testing Frontier Large Language Models' Physics Literacy in Parallel Physical Worlds","primary_cat":"cs.LG","submitted_at":"2026-06-30T23:52:15+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Introduces an auditable four-stage diagnostic for LLM physics reasoning in novel frameworks and applies it to three parallel worlds, yielding pass rates of 6/15, 6/15, and 0/15 on frontier models with noted qualitative-quantitative asymmetry.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.00254","ref_index":5,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Query-Centric Optimization of AI Workflows via Approximate Query Processing and Proxy Models","primary_cat":"cs.DB","submitted_at":"2026-06-30T23:05:59+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":4.0,"formal_verification":"none","one_line_summary":"Query-centric AQP and proxy-model strategies reduce expensive model calls by 60-90% with under 10% error on TPC-DS and LLM tasks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.00208","ref_index":5,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"SLIM-RL: Risk-Budgeted Random-Masking RL for Diffusion LLMs Without Trajectory Slicing","primary_cat":"cs.CL","submitted_at":"2026-06-30T21:38:46+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"SLIM-RL matches or exceeds TraceRL performance on MATH500, GSM8K, MBPP and HumanEval for diffusion LLMs by risk-budgeted random-masking RL without trajectory slicing.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.00162","ref_index":63,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"FRAME: Learning the Adaptation Domain with a Mixture of Fractional-Fourier Experts","primary_cat":"cs.LG","submitted_at":"2026-06-30T20:39:55+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"FRAME adds a learnable fractional-Fourier order per expert in a MoE-LoRA setup so that low-rank updates are placed in the domain where they are most compact, yielding gains over fixed-domain baselines on LLaMA-3.1-8B and Qwen2.5-7B.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.31813","ref_index":56,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Geometry-Preserving Orthonormal Initialization for Low-Rank Adaptation in RLVR","primary_cat":"cs.LG","submitted_at":"2026-06-30T15:27:54+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Orthonormal initialization for LoRA in RLVR achieves the minimal gap to full fine-tuning, stabilizes training, and outperforms standard LoRA and prior variants on mathematical reasoning benchmarks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.31779","ref_index":119,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Bridging the Gap Between Latent and Explicit Reasoning with Looped Transformers","primary_cat":"cs.LG","submitted_at":"2026-06-30T14:58:53+00:00","verdict":"ACCEPT","verdict_confidence":"HIGH","novelty_score":6.0,"formal_verification":"none","one_line_summary":"A looped padded Transformer with parallel gold-CoT cross-entropy supervision matches explicit CoT accuracy at 3B scale and is 2.5–6.9× faster in the thought phase.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.31748","ref_index":62,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Addressing Over-Refusal in LLMs with Competing Rewards","primary_cat":"cs.LG","submitted_at":"2026-06-30T14:38:49+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"SEAR trains one LLM via adversarial process rewards to explore harmful reasoning paths but flip to safe outputs, reducing over-refusal while preserving safety.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.31717","ref_index":12,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Nonlinearity-Aware LoRA: Structured Gate Adaptation under Low-Rank Constraints","primary_cat":"cs.LG","submitted_at":"2026-06-30T14:21:51+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"NA-LoRA introduces derivative-based temporal-importance masks and activation-specific step scaling to LoRA to reduce selection misalignment in self-gated FFNs, with reported gains on language and vision-language fine-tuning tasks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.31693","ref_index":43,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"ShopX: A Foundation Model for Intent-to-Item Fulfillment in Agentic Shopping","primary_cat":"cs.IR","submitted_at":"2026-06-30T14:05:28+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"A single LLM trained to emit semantic item codes can fulfill complex shopping intents with fewer tool hand-offs, improving multi-turn follow-up on Taobao-derived tasks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.31608","ref_index":112,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"CLExEval: A Human-in-the-Loop Framework for Qualitative Evaluation of LLM Clinical Reasoning","primary_cat":"cs.CL","submitted_at":"2026-06-30T12:56:42+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"CLExEval introduces a human-annotated evaluation framework on 40 rare cases that identifies verbosity bias, hidden knowledge paradox, and 68.6% reasoning-to-output mismatch in LLMs while showing LLM-as-a-Judge overestimates reliability.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.31580","ref_index":6,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"LASER: Load-Aware Serving with Early-Exit for Reasoning LLMs at the Edge","primary_cat":"cs.DC","submitted_at":"2026-06-30T12:35:48+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"LASER reduces edge LLM serving latency by 17-38% and improves SLO satisfaction by 3-6% via load-aware adaptive early-exit thresholds and difficulty-aware budget pre-allocation, with 1% average accuracy cost.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.31551","ref_index":11,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"AutoTrainess: Teaching Language Models to Improve Language Models Autonomously","primary_cat":"cs.CL","submitted_at":"2026-06-30T12:09:51+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"AutoTrainess exposes training operations via agent-computer interfaces and outperforms CLI-only baselines on PostTrainBench with scores of 26.94 vs 23.21 for GPT-5.4 and similar gains on other models.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.31519","ref_index":45,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"RaBitQCache: Rotated Binary Quantization for KVCache in Long Context LLM Inference","primary_cat":"cs.LG","submitted_at":"2026-06-30T11:32:14+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"RaBitQCache proposes rotated binary quantization with binary-INT4 arithmetic for unbiased attention weight estimation in long-context LLMs, enabling adaptive Top-p retrieval and hardware optimizations.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.31315","ref_index":13,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"BlockPilot: Instance-Adaptive Policy Learning for Diffusion-based Speculative Decoding","primary_cat":"cs.CL","submitted_at":"2026-06-30T08:24:05+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"BlockPilot is an instance-adaptive policy that predicts optimal block size from the prefilling representation for diffusion speculative decoding, reporting 5.92 acceptance length and 4.20x speedup on Qwen3-4B.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.31092","ref_index":15,"ref_count":2,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Fora: From Weight-Space to Function-Space Protection in Capability-Preserving Fine-Tuning","primary_cat":"cs.LG","submitted_at":"2026-06-30T03:28:56+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":7.0,"formal_verification":"none","one_line_summary":"FORA uses function-space projectors from activation covariances to enable capability-preserving fine-tuning of LLMs, outperforming weight-space methods on preservation tasks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.31048","ref_index":6,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Knowledge Distillation from Large Reasoning Models to Compact Student Models: A Case Study on the John O Bryan Mathematics Competition","primary_cat":"cs.LG","submitted_at":"2026-06-30T02:34:45+00:00","verdict":"CONDITIONAL","verdict_confidence":"LOW","novelty_score":4.0,"formal_verification":"none","one_line_summary":"Distilling CoT from DeepSeek-R1 to Qwen2.5-7B on competition problems yields 4.76 pp accuracy gain to 69.43% and 73.1% on MATH-500, with accuracy falling as response length decreases.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.30923","ref_index":14,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Behavior Cloning is Not All You Need: The Optimality of On-Policy Distillation for Noisy Expert Feedback","primary_cat":"cs.LG","submitted_at":"2026-06-29T21:18:21+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":8.0,"formal_verification":"none","one_line_summary":"Noisy expert imitation learning requires exponential samples for offline methods but polynomial for a variant of on-policy distillation under a noise condition.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.30852","ref_index":3,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"When Does Learning to Stop Help? A Cost-Aware Study of Early Exits in Reasoning Models","primary_cat":"cs.AI","submitted_at":"2026-06-29T19:33:42+00:00","verdict":"ACCEPT","verdict_confidence":"HIGH","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Under matched lost-correct risk, learned early exits beat calibrated scalar exits on free-form math, lose on multiple-choice, and neither certifies savings on small hard sets; serving regime can reverse the savings.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.30790","ref_index":73,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Indi-RomCoM: Code-Mixed Benchmark for Evaluating LLMs on Romanized Indic-English Instructions","primary_cat":"cs.CL","submitted_at":"2026-06-29T18:19:24+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"Introduces Indi-RomCoM benchmark for evaluating LLMs on Romanized code-mixed Indic-English instructions across seven tasks, four languages, and three mixing levels.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.30789","ref_index":1,"ref_count":2,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Predictable GRPO: A Closed-Form Model of Training Dynamics","primary_cat":"cs.LG","submitted_at":"2026-06-29T18:19:09+00:00","verdict":"UNVERDICTED","verdict_confidence":"MODERATE","novelty_score":7.0,"formal_verification":"none","one_line_summary":"GRPO updates reduce to a damped oscillator whose mass, damping, and stiffness are fixed by optimizer hyperparameters plus one measured curvature scale, subsuming single-exponential saturation while adding inertial slow-start and group-size predictions.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.30627","ref_index":3,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Pessimism's Paradox: Conservative Offline Training Amplifies Reward Hacking During Online Adaptation in Reasoning Models","primary_cat":"cs.LG","submitted_at":"2026-06-29T17:56:03+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Higher conservatism in offline DPO training of Qwen3-14B monotonically increases reward-hacking damage (Goodhart gap AUGC) during online adaptation on GSM8K.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.30602","ref_index":20,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"MESA: Prioritizing Vulnerable Communication Channels for Securing Multi-Agent Systems","primary_cat":"cs.CR","submitted_at":"2026-06-29T17:40:45+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"MESA ranks MAS communication edges by vulnerability via graph-theoretic metrics and dynamic probes, achieving mean Spearman ρ=+0.60 correlation with empirical per-edge attack success and 3x interception gain when monitoring the top 10%.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.30445","ref_index":8,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"When Does Online Imitation Learning Help in LLM Post-Training? The Role of (Non-)Realizability Beyond Horizon","primary_cat":"cs.LG","submitted_at":"2026-06-29T15:17:42+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"Online IL overcomes an information-theoretic bottleneck that offline IL faces in non-realizable settings even at horizon 1, under a new structural characterization of reward-relative misspecification.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.30360","ref_index":9,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"On the Vulnerability of Parameter-Level Defenses to Model Merging","primary_cat":"cs.LG","submitted_at":"2026-06-29T14:26:13+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Parameter-level defenses for model merging are vulnerable to Anchor-Guided Attack because protected weights are dominated by the pretrained model, and a new defense ARF is introduced to counter it.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.30107","ref_index":12,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Structural Certification for Reliable Physical Design with Language Models","primary_cat":"cs.AI","submitted_at":"2026-06-29T10:43:57+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"PHACT uses a propose-certify loop with deterministic derivation from fixed inputs to achieve zero false certifications in 80 adversarial trials across two models and temperatures.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.29982","ref_index":10,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Beyond Uniform Experts: Cost-Aware Expert Execution for Efficient Multi-Device MoE Inference","primary_cat":"cs.DC","submitted_at":"2026-06-29T08:57:59+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"CAEE reduces MoE inference latency 8-18% on 671B DeepSeek-R1 by cost-aware expert pruning and low-overhead compensation while keeping accuracy drop under 1%.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.29758","ref_index":27,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"PS-PPO: Prefix-Sampling PPO for Critic-Free RLHF","primary_cat":"cs.LG","submitted_at":"2026-06-29T04:04:32+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":7.0,"formal_verification":"none","one_line_summary":"PS-PPO samples a per-trajectory cutoff and importance-weights truncated gradients, preserving the full critic-free update in expectation while cutting RLHF training compute and memory.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.29712","ref_index":6,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Why Struggle with Continuous Latents? Interpretable Discrete Latent Reasoning via Rendered Compression","primary_cat":"cs.CL","submitted_at":"2026-06-29T02:34:52+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"DLR creates discrete latent tokens from rendered CoT images via clustering, enabling up to 20x compression and interpretable trajectories that outperform continuous latent baselines on reasoning tasks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.29481","ref_index":106,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"To Reason or to Fabricate: Reasoning Without Shortcuts via Hint-Anchored Pairwise Aggregation","primary_cat":"cs.CL","submitted_at":"2026-06-28T16:21:04+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"HIPPO is a new RL framework that uses hint-anchored pairwise aggregation to distinguish and promote authentic reasoning deduction in LLMs instead of shortcut memorization from data overlap.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.29476","ref_index":28,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"CRAFT: Counterfactual Credit Assignment from Free Sibling Rollouts for Self-Distilled Agentic Reinforcement Learning","primary_cat":"cs.LG","submitted_at":"2026-06-28T16:11:47+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"CRAFT is a three-pillar credit assignment scheme that uses counterfactual token importance from GRPO sibling rollouts to provide signed per-token distillation signals in self-distilled agentic RL.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.29354","ref_index":44,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"When LLMs Develop Languages: Symbolic Communication for Efficient Multi-Agent Reasoning","primary_cat":"cs.AI","submitted_at":"2026-06-28T12:02:42+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"CLSR lets LLM agents evolve and route symbolic languages that reduce generated tokens by 3-6x versus chain-of-thought while keeping accuracy on benchmarks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.29278","ref_index":2,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"The Complexity Ceiling Benchmark: A Multi-Domain Evaluation of Sequential Reasoning Under Depth Scaling","primary_cat":"cs.AI","submitted_at":"2026-06-28T08:56:34+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"The Complexity Ceiling Benchmark demonstrates geometric per-step decay in LLM sequential reasoning with domain-specific performance ceilings and introduces a trace metric showing incorrect intermediate steps in some correct final answers.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.29270","ref_index":6,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Minority Sentinel: When to Overturn Majority Voting in Multi-Agent LLM Debates","primary_cat":"cs.MA","submitted_at":"2026-06-28T08:37:22+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"Minority Sentinel uses a LightGBM model on debate fingerprints to overturn majority votes in LLM debates with 81.2% flip precision and positive net gain on six benchmarks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.29223","ref_index":16,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Depth Exploration for LLM Decoding","primary_cat":"cs.LG","submitted_at":"2026-06-28T06:22:09+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"DEX replaces single-depth selection with parallel exploration over multiple candidate depths, committing the final-depth token while collapsing reusable states to reduce per-token computation.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.29215","ref_index":33,"ref_count":2,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Multi-Block Diffusion Language Models","primary_cat":"cs.LG","submitted_at":"2026-06-28T05:53:45+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"MBD-LMs raise average tokens per forward pass from 3.47 to 6.19 (and to 9.34 with DMax) via multi-block teacher forcing and optimized parallel decoding while holding or slightly improving accuracy on math and code tasks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.29164","ref_index":3,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Invariant Reasoning Directions in Latent Trajectories of Language Models","primary_cat":"cs.LG","submitted_at":"2026-06-28T02:59:08+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"TILR identifies low-rank invariant subspaces from contrastive latent trajectory differences in LLMs and constrains interventions to them, improving paraphrase consistency by ~10% and reducing variance by up to 50%.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.29150","ref_index":7,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Flow Reasoning Models: Scaling Reasoning Through Iterative Self-Refinement","primary_cat":"cs.AI","submitted_at":"2026-06-28T02:10:36+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":7.0,"formal_verification":"none","one_line_summary":"Flow models reach 99.2% Sudoku accuracy in 7 passes and 96.1% on out-of-distribution Sudoku-Extreme by selecting dynamically stable candidates and training with self-conditioning plus DPO to avoid failed outputs.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.29094","ref_index":10,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"DiLaServe: High SLO Attainment Serving for Diffusion Language Models","primary_cat":"cs.LG","submitted_at":"2026-06-27T21:21:53+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"DiLaServe improves SLO attainment for diffusion language models by up to 56.6 percentage points and reduces latency by up to 46% with less than 1% accuracy drop via deadline-aware scheduling and dynamic reconfiguration.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.28831","ref_index":5,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"HARD-KV: Head-Adaptive Regularization for Decoding-time KV Compression","primary_cat":"cs.LG","submitted_at":"2026-06-27T09:36:37+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"HARD-KV bridges dynamic head-adaptive KV cache compression with static inference engine constraints via Cascade Cache and Logits Calibration, reporting up to 2x throughput gains on long-context math benchmarks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.28733","ref_index":82,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Agentic Abstention: Do Agents Know When to Stop Instead of Act?","primary_cat":"cs.AI","submitted_at":"2026-06-27T04:49:37+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"LLM agents often fail to abstain at the right time in uncertain multi-turn tasks, and the CONVOLVE context engineering method raises timely abstention rates on WebShop from 26.7 to 57.4 without parameter updates.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.28707","ref_index":59,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"BV-Blend: Uncertainty-Weighted Historical Baselines for Stable Critic-Free RL with Verifiable Rewards","primary_cat":"cs.AI","submitted_at":"2026-06-27T03:25:53+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":4.0,"formal_verification":"none","one_line_summary":"BV-Blend blends prompt-local and semantic-cluster historical reward statistics via SEM-derived weights to stabilize critic-free RL advantage estimation.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.28661","ref_index":21,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"When More Sampling Hurts: The Modal Ceiling and Correlation Ceiling of Test-Time Scaling","primary_cat":"cs.LG","submitted_at":"2026-06-27T00:37:33+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Test-time sampling improves coverage but stalls at modal and correlation ceilings for answer selection, with the effective number of samples as the practical limit.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.28551","ref_index":47,"ref_count":2,"confidence":0.98,"is_internal_anchor":true,"paper_title":"DataComp-VLM: Improved Open Datasets for Vision-Language Models","primary_cat":"cs.CV","submitted_at":"2026-06-26T19:11:29+00:00","verdict":null,"verdict_confidence":null,"novelty_score":null,"formal_verification":null,"one_line_summary":null,"context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.28301","ref_index":7,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"VGB for Masked Diffusion Model: Efficient Test-time Scaling for Reward Satisfaction and Sample Editing","primary_cat":"cs.LG","submitted_at":"2026-06-26T17:47:09+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"MDM-VGB augments masked diffusion with backtracking-style reward-guided remasking to achieve quadratic-complexity high-reward generation and sample editing, with proofs of noise robustness.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.28471","ref_index":7,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Data and Evaluation Closed-Loop for Model Capability Enhancement","primary_cat":"cs.AI","submitted_at":"2026-06-26T14:45:57+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Proposes capability slices with dual taxonomies and mapping rules to form a closed loop converting benchmark failures into targeted data interventions, validated via two opposing case studies on BBH and math reasoning.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.27752","ref_index":6,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"PerturbCellRL: Verifier-Guided Reinforcement Learning for Single-Cell Perturbation Prediction","primary_cat":"cs.LG","submitted_at":"2026-06-26T06:15:04+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"PerturbCellRL is a reinforcement learning framework that post-trains single-cell transcriptomic generators using verifier rewards for improved biological consistency in perturbation predictions.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.27739","ref_index":2,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"The Weakest Link Tells It All: Outcome-Supervised Process Reward Modeling via Learnable Credit Assignment","primary_cat":"cs.LG","submitted_at":"2026-06-26T05:38:08+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"LCA frames outcome-supervised PRM training as MIL, introduces SWS pooling for dependent steps, proves Bayes consistency under mild assumptions, and reports consistent gains over prior outcome-supervised baselines.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.28430","ref_index":1,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Building to the Test: Coding Agents Deliver What You Check, Not What You Requested","primary_cat":"cs.SE","submitted_at":"2026-06-26T01:23:27+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Coding agents build libraries that pass a hidden 222-test oracle by producing minimal demos holding only the tested behavior, while leaving the requested functionality unfinished or absent when the oracle is unavailable.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.27632","ref_index":7,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Yuvion LLM: An Adversarially-Aware Large Language Model for Content And AI Safety","primary_cat":"cs.CL","submitted_at":"2026-06-26T01:12:02+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Yuvion LLM applies adversarially aware training and introduces the YLRE benchmark set, claiming superior safety robustness over larger models on multiple tasks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.27617","ref_index":12,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Masked Language Flow Models","primary_cat":"cs.CL","submitted_at":"2026-06-26T00:16:40+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"MLFMs combine masking with continuous flows to scale flow-based language models to reasoning and instruction-following tasks on GSM8K and MT-Bench.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.27550","ref_index":15,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"EntMTP: Accelerating LLM Inference with Entropy Guided Multi Token Prediction","primary_cat":"cs.CL","submitted_at":"2026-06-25T20:54:27+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"EntMTP is a training-free entropy-guided scheduler for multi-token prediction that dynamically selects from task-specific Pareto-optimal trees to accelerate LLM inference by up to 1.36x on benchmarks without quality loss.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.27359","ref_index":8,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"When are likely answers right? On Sequence Probability and Correctness in LLMs","primary_cat":"stat.ML","submitted_at":"2026-06-25T17:58:02+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Higher sequence probability predicts correctness across different answers in a dataset but does not reliably improve accuracy when decoding methods or hyperparameters are changed, nor does it indicate correctness for repeated responses to one prompt.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.27047","ref_index":9,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"NuclearQAv2: A Structured Benchmark for Evaluating Domain-Science Competence in Large Language Models","primary_cat":"cs.CL","submitted_at":"2026-06-25T13:52:16+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"NuclearQAv2 is a hybrid-constructed benchmark dataset for evaluating LLM competence in nuclear engineering knowledge using three question types.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.26969","ref_index":41,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Einstein World Models","primary_cat":"cs.AI","submitted_at":"2026-06-25T12:42:04+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Einstein World Models integrate visual rollouts from a callable world-module into LLM reasoning traces to support complex thought beyond language.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null}],"limit":100,"offset":0}}