{"work":{"id":"e6b75ad5-2877-4168-97c8-710407094d20","openalex_id":"https://openalex.org/W4403625924","doi":"10.1016/j.artmed.2024.103001","arxiv_id":"2501.12948","raw_key":null,"title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","authors":null,"authors_text":"DeepSeek-AI","year":2025,"venue":"cs.CL","abstract":"General reasoning represents a long-standing and formidable challenge in artificial intelligence. Recent breakthroughs, exemplified by large language models (LLMs) and chain-of-thought prompting, have achieved considerable success on foundational reasoning tasks. However, this success is heavily contingent upon extensive human-annotated demonstrations, and models' capabilities are still insufficient for more complex problems. Here we show that the reasoning abilities of LLMs can be incentivized through pure reinforcement learning (RL), obviating the need for human-labeled reasoning trajectories. The proposed RL framework facilitates the emergent development of advanced reasoning patterns, such as self-reflection, verification, and dynamic strategy adaptation. Consequently, the trained model achieves superior performance on verifiable tasks such as mathematics, coding competitions, and STEM fields, surpassing its counterparts trained via conventional supervised learning on human demonstrations. Moreover, the emergent reasoning patterns exhibited by these large-scale models can be systematically harnessed to guide and enhance the reasoning capabilities of smaller models.","external_url":"https://arxiv.org/abs/2501.12948","cited_by_count":24,"metadata_source":"pith","metadata_fetched_at":"2026-08-05T02:28:24.338817+00:00","pith_arxiv_id":"2501.12948","created_at":"2026-05-08T17:13:38.764111+00:00","updated_at":"2026-08-05T02:28:24.338817+00:00","title_quality_ok":true,"display_title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","render_title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning"},"hub":{"state":{"work_id":"e6b75ad5-2877-4168-97c8-710407094d20","tier":"mega_hub","tier_reason":"1,000+ Pith inbound or 100,000+ external citations","pith_inbound_count":1939,"external_cited_by_count":24,"distinct_field_count":43,"first_pith_cited_at":"2024-05-01T15:59:00+00:00","last_pith_cited_at":"2026-07-09T16:17:10+00:00","author_build_status":"needed","summary_status":"needed","contexts_status":"needed","graph_status":"needed","ask_index_status":"needed","reader_status":"needed","recognition_status":"needed","updated_at":"2026-08-23T12:19:19.781420+00:00","tier_text":"mega_hub"},"tier":"mega_hub","role_counts":[{"context_role":"background","n":272},{"context_role":"method","n":48},{"context_role":"baseline","n":28},{"context_role":"dataset","n":12},{"context_role":"other","n":6}],"polarity_counts":[{"context_polarity":"background","n":259},{"context_polarity":"use_method","n":45},{"context_polarity":"baseline","n":28},{"context_polarity":"unclear","n":18},{"context_polarity":"use_dataset","n":11},{"context_polarity":"support","n":4},{"context_polarity":"extend","n":1}],"runs":{"ask_index":{"job_type":"ask_index","status":"succeeded","result":{"title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","claims":[{"claim_text":"General reasoning represents a long-standing and formidable challenge in artificial intelligence. Recent breakthroughs, exemplified by large language models (LLMs) and chain-of-thought prompting, have achieved considerable success on foundational reasoning tasks. However, this success is heavily contingent upon extensive human-annotated demonstrations, and models' capabilities are still insufficient for more complex problems. Here we show that the reasoning abilities of LLMs can be incentivized through pure reinforcement learning (RL), obviating the need for human-labeled reasoning trajectorie","claim_type":"abstract","evidence_strength":"source_metadata"}],"why_cited":"Pith tracks DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning because it crossed a citation-hub threshold.","role_counts":[]},"error":null,"updated_at":"2026-05-13T17:53:36.998847+00:00"},"author_expand":{"job_type":"author_expand","status":"succeeded","result":{"authors_linked":[{"id":"b3d3bc38-c7e6-4554-ab1a-a4b32a8299c8","orcid":null,"display_name":"DeepSeek-AI"}]},"error":null,"updated_at":"2026-05-13T17:24:04.935515+00:00"},"context_extract":{"job_type":"context_extract","status":"succeeded","result":{"enqueued_papers":25},"error":null,"updated_at":"2026-05-13T17:53:36.992694+00:00"},"graph_features":{"job_type":"graph_features","status":"succeeded","result":{"co_cited":[{"title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","work_id":"c5006563-f3ec-438a-9e35-b7b484f34828","shared_citers":241},{"title":"Qwen3 Technical Report","work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e","shared_citers":211},{"title":"Proximal Policy Optimization Algorithms","work_id":"240c67fe-d14d-4520-91c1-38a4e272ca19","shared_citers":137},{"title":"GPT-4 Technical Report","work_id":"b928e041-6991-4c08-8c81-0359e4097c7b","shared_citers":114},{"title":"The Llama 3 Herd of Models","work_id":"1549a635-88af-4ac1-acfe-51ae7bb53345","shared_citers":114},{"title":"DAPO: An Open-Source LLM Reinforcement Learning System at Scale","work_id":"64019d00-0b11-4bbd-b173-b46c8fad0157","shared_citers":113},{"title":"Training Verifiers to Solve Math Word Problems","work_id":"acab1aa8-b4d6-40e0-a3ee-25341701dca2","shared_citers":104},{"title":"OpenAI o1 System Card","work_id":"68d3c334-0fc9-49e3-b7b0-a69afae933e2","shared_citers":103},{"title":"Qwen2.5 Technical Report","work_id":"d8432992-4980-4a81-85c7-9fa2c2b87f85","shared_citers":77},{"title":"GPT-4o System Card","work_id":"f37bf1c7-4964-4e56-9762-d20da8d9009f","shared_citers":73},{"title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","work_id":"008df105-2fdd-45d8-857a-8e35868aecb6","shared_citers":72},{"title":"Qwen2.5-VL Technical Report","work_id":"69dffacb-bfe8-442d-be86-48624c60426f","shared_citers":72},{"title":"DeepSeek-V3 Technical Report","work_id":"57d2791d-2219-4c31-a077-afc04b12a75c","shared_citers":67},{"title":"Evaluating Large Language Models Trained on Code","work_id":"042493e9-b26f-4b4e-bbde-382072ca9b08","shared_citers":61},{"title":"Measuring Mathematical Problem Solving With the MATH Dataset","work_id":"50652ac6-fb7c-4675-a2c2-159c241feb17","shared_citers":56},{"title":"Group Sequence Policy Optimization","work_id":"3a98b53b-9f52-4d95-adf7-89353c0a9a65","shared_citers":51},{"title":"Qwen3-VL Technical Report","work_id":"1fe243aa-e3c0-4da6-b391-4cbcfc88d5c0","shared_citers":51},{"title":"Kimi k1.5: Scaling Reinforcement Learning with LLMs","work_id":"bff96ab1-bd6a-4585-be23-74fdb51969c7","shared_citers":47},{"title":"Understanding R1-Zero-Like Training: A Critical Perspective","work_id":"ec354f3b-9484-4a0c-94c8-92d4d0260835","shared_citers":45},{"title":"Scaling LLM Test-Time Compute Optimally can be More Effective than Scaling Model Parameters","work_id":"a8d50b24-bdf5-46ed-bc4f-2927dfd81f1d","shared_citers":41},{"title":"LLaMA: Open and Efficient Foundation Language Models","work_id":"c018fc23-6f3f-4035-9d02-28a2173b2b9d","shared_citers":40},{"title":"LiveCodeBench: Holistic and Contamination Free Evaluation of Large Language Models for Code","work_id":"ea9e51ce-1e75-4182-92d8-4d25f70d2ee4","shared_citers":35},{"title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","work_id":"8abcfe4f-e0fb-44b7-9123-448fac95f90a","shared_citers":35},{"title":"Self-Consistency Improves Chain of Thought Reasoning in Language Models","work_id":"8c6d5a6b-b5cc-4105-9c84-9c34bb9375bb","shared_citers":34}],"time_series":[{"n":2,"year":2024},{"n":34,"year":2025},{"n":563,"year":2026}]},"error":null,"updated_at":"2026-05-13T17:25:54.879041+00:00"},"identity_refresh":{"job_type":"identity_refresh","status":"succeeded","result":{"fixed":1,"items":[{"title":"Qwen3 Technical Report","work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e","resolver":"local_arxiv","confidence":0.98,"old_work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e"}],"errors":[],"attempted":1},"error":null,"updated_at":"2026-05-13T17:53:36.131917+00:00"},"reader_index":{"job_type":"reader_index","status":"succeeded","result":{"note":"annotated reader requires full-text/OA fetch; shell is wired for mega hubs","status":"reader queued"},"error":null,"updated_at":"2026-05-19T02:51:43.032474+00:00"},"recognition_alignment":{"job_type":"recognition_alignment","status":"succeeded","result":{"modules":["IndisputableMonolith.Cognition.AnalogicalReasoningFromJCost","IndisputableMonolith.Cognition.AnimalZComplexityBound","IndisputableMonolith.Foundation.VoiceForcing","IndisputableMonolith.Education.PedagogyModelsFromConfigDim","IndisputableMonolith.Cosmology.CosmologicalConstant","IndisputableMonolith.Core.ConstantsAndPatterns","IndisputableMonolith.Materials.RoomTSuperconductorCandidate","IndisputableMonolith.Common.CanonicalJBand"],"query_chars":1270},"error":null,"updated_at":"2026-05-19T02:51:43.029049+00:00"},"role_polarity":{"job_type":"role_polarity","status":"succeeded","result":{"title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","claims":[{"claim_text":"General reasoning represents a long-standing and formidable challenge in artificial intelligence. Recent breakthroughs, exemplified by large language models (LLMs) and chain-of-thought prompting, have achieved considerable success on foundational reasoning tasks. However, this success is heavily contingent upon extensive human-annotated demonstrations, and models' capabilities are still insufficient for more complex problems. Here we show that the reasoning abilities of LLMs can be incentivized through pure reinforcement learning (RL), obviating the need for human-labeled reasoning trajectorie","claim_type":"abstract","evidence_strength":"source_metadata"}],"why_cited":"Pith tracks DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning because it crossed a citation-hub threshold.","role_counts":[]},"error":null,"updated_at":"2026-05-13T17:53:36.995986+00:00"},"summary_claims":{"job_type":"summary_claims","status":"succeeded","result":{"title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","claims":[{"claim_text":"General reasoning represents a long-standing and formidable challenge in artificial intelligence. Recent breakthroughs, exemplified by large language models (LLMs) and chain-of-thought prompting, have achieved considerable success on foundational reasoning tasks. However, this success is heavily contingent upon extensive human-annotated demonstrations, and models' capabilities are still insufficient for more complex problems. Here we show that the reasoning abilities of LLMs can be incentivized through pure reinforcement learning (RL), obviating the need for human-labeled reasoning trajectorie","claim_type":"abstract","evidence_strength":"source_metadata"}],"why_cited":"Pith tracks DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning because it crossed a citation-hub threshold.","role_counts":[]},"error":null,"updated_at":"2026-05-13T17:25:52.711836+00:00"}},"summary":{"title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","claims":[{"claim_text":"General reasoning represents a long-standing and formidable challenge in artificial intelligence. Recent breakthroughs, exemplified by large language models (LLMs) and chain-of-thought prompting, have achieved considerable success on foundational reasoning tasks. However, this success is heavily contingent upon extensive human-annotated demonstrations, and models' capabilities are still insufficient for more complex problems. Here we show that the reasoning abilities of LLMs can be incentivized through pure reinforcement learning (RL), obviating the need for human-labeled reasoning trajectorie","claim_type":"abstract","evidence_strength":"source_metadata"}],"why_cited":"Pith tracks DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning because it crossed a citation-hub threshold.","role_counts":[]},"graph":{"co_cited":[{"title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","work_id":"c5006563-f3ec-438a-9e35-b7b484f34828","shared_citers":241},{"title":"Qwen3 Technical Report","work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e","shared_citers":211},{"title":"Proximal Policy Optimization Algorithms","work_id":"240c67fe-d14d-4520-91c1-38a4e272ca19","shared_citers":137},{"title":"GPT-4 Technical Report","work_id":"b928e041-6991-4c08-8c81-0359e4097c7b","shared_citers":114},{"title":"The Llama 3 Herd of Models","work_id":"1549a635-88af-4ac1-acfe-51ae7bb53345","shared_citers":114},{"title":"DAPO: An Open-Source LLM Reinforcement Learning System at Scale","work_id":"64019d00-0b11-4bbd-b173-b46c8fad0157","shared_citers":113},{"title":"Training Verifiers to Solve Math Word Problems","work_id":"acab1aa8-b4d6-40e0-a3ee-25341701dca2","shared_citers":104},{"title":"OpenAI o1 System Card","work_id":"68d3c334-0fc9-49e3-b7b0-a69afae933e2","shared_citers":103},{"title":"Qwen2.5 Technical Report","work_id":"d8432992-4980-4a81-85c7-9fa2c2b87f85","shared_citers":77},{"title":"GPT-4o System Card","work_id":"f37bf1c7-4964-4e56-9762-d20da8d9009f","shared_citers":73},{"title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","work_id":"008df105-2fdd-45d8-857a-8e35868aecb6","shared_citers":72},{"title":"Qwen2.5-VL Technical Report","work_id":"69dffacb-bfe8-442d-be86-48624c60426f","shared_citers":72},{"title":"DeepSeek-V3 Technical Report","work_id":"57d2791d-2219-4c31-a077-afc04b12a75c","shared_citers":67},{"title":"Evaluating Large Language Models Trained on Code","work_id":"042493e9-b26f-4b4e-bbde-382072ca9b08","shared_citers":61},{"title":"Measuring Mathematical Problem Solving With the MATH Dataset","work_id":"50652ac6-fb7c-4675-a2c2-159c241feb17","shared_citers":56},{"title":"Group Sequence Policy Optimization","work_id":"3a98b53b-9f52-4d95-adf7-89353c0a9a65","shared_citers":51},{"title":"Qwen3-VL Technical Report","work_id":"1fe243aa-e3c0-4da6-b391-4cbcfc88d5c0","shared_citers":51},{"title":"Kimi k1.5: Scaling Reinforcement Learning with LLMs","work_id":"bff96ab1-bd6a-4585-be23-74fdb51969c7","shared_citers":47},{"title":"Understanding R1-Zero-Like Training: A Critical Perspective","work_id":"ec354f3b-9484-4a0c-94c8-92d4d0260835","shared_citers":45},{"title":"Scaling LLM Test-Time Compute Optimally can be More Effective than Scaling Model Parameters","work_id":"a8d50b24-bdf5-46ed-bc4f-2927dfd81f1d","shared_citers":41},{"title":"LLaMA: Open and Efficient Foundation Language Models","work_id":"c018fc23-6f3f-4035-9d02-28a2173b2b9d","shared_citers":40},{"title":"LiveCodeBench: Holistic and Contamination Free Evaluation of Large Language Models for Code","work_id":"ea9e51ce-1e75-4182-92d8-4d25f70d2ee4","shared_citers":35},{"title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","work_id":"8abcfe4f-e0fb-44b7-9123-448fac95f90a","shared_citers":35},{"title":"Self-Consistency Improves Chain of Thought Reasoning in Language Models","work_id":"8c6d5a6b-b5cc-4105-9c84-9c34bb9375bb","shared_citers":34}],"time_series":[{"n":2,"year":2024},{"n":34,"year":2025},{"n":563,"year":2026}]},"authors":[{"id":"b3d3bc38-c7e6-4554-ab1a-a4b32a8299c8","orcid":null,"display_name":"DeepSeek-AI","source":"manual","import_confidence":0.72}]},"citers":{"total":1939,"items":[{"citing_arxiv_id":"2607.08643","ref_index":5,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"BiSCo-LLM: Lookup-Free Binary Spherical Coding for Extreme Low-Bit Large Language Model Compression","primary_cat":"cs.LG","submitted_at":"2026-07-09T16:17:10+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"BiSCo-LLM achieves near-FP16 accuracy on Qwen3-8B at ~2 bits/weight using codebook-free binary spherical codes with residual coding and category-wise recovery distillation.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.08572","ref_index":4,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Switch-Reasoner: Learn When to Think in Multitask Mixtures via Reinforcement Learning","primary_cat":"cs.CV","submitted_at":"2026-07-09T15:04:05+00:00","verdict":"CONDITIONAL","verdict_confidence":"HIGH","novelty_score":5.0,"formal_verification":"none","one_line_summary":"A GRPO framework that treats thinking as a tool call and uses dual-level regulation so multimodal models learn when to reason versus answer directly.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.08375","ref_index":14,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"WCog-VLA: A Dual-Level World-Cognitive Vision-Language-Action Model for End-to-End Autonomous Driving","primary_cat":"cs.CV","submitted_at":"2026-07-09T11:49:57+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"WCog-VLA couples Game-CoT semantic reasoning with an aligned decoupled diffusion transformer to generate joint multi-agent trajectories and reaches 92.9 PDMS on NAVSIM.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.08374","ref_index":39,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Large-Language-Models-as-a-Judge in Theory-Agnostic Adaptive Metric-Alignment for Prototypical Networks in Personality Recognition","primary_cat":"cs.CL","submitted_at":"2026-07-09T11:49:14+00:00","verdict":"CONDITIONAL","verdict_confidence":"HIGH","novelty_score":6.0,"formal_verification":"none","one_line_summary":"JAM discovers theory-invariant pseudo-facets via attention-pooled graph prototypical networks, Cross-Theory Harmonization, and LLM-as-a-Judge, improving cross-framework balanced accuracy on Essays and Kaggle datasets.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.08268","ref_index":10,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Different Teachers, Different Capabilities: Sub-1B On-Device Distillation for Structured Text Enrichment","primary_cat":"cs.AI","submitted_at":"2026-07-09T09:10:49+00:00","verdict":"CONDITIONAL","verdict_confidence":"HIGH","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Distilling an 8B reasoning teacher into a 0.6B student recovers most summary quality at ~50× speed, but teacher type—not scale alone—determines which capabilities transfer.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.08124","ref_index":27,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"TTHE: Test-Time Harness Evolution","primary_cat":"cs.SE","submitted_at":"2026-07-09T05:53:39+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"An LLM agent can improve itself at test time by rewriting its surrounding executable harness from unlabeled traces, using only proxy signals and a frozen model.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.08116","ref_index":8,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"MORES: Mobile Reasoning-as-a-Service via Distributed LLM Inference-Time Scaling","primary_cat":"cs.NI","submitted_at":"2026-07-09T05:24:31+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":5.5,"formal_verification":"none","one_line_summary":"A device–server split of recurrent latent LLM reasoning plus semantic MoE-SAC scheduling yields about 18% higher simulated system throughput than plain SAC under energy, recurrence, and latency budgets.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.07993","ref_index":3,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Hallucination Self-Play: Bootstrapping Reinforced Detector via Evolved Generator","primary_cat":"cs.CL","submitted_at":"2026-07-08T23:54:36+00:00","verdict":"CONDITIONAL","verdict_confidence":"HIGH","novelty_score":7.0,"formal_verification":"none","one_line_summary":"Hallucination Self-Play co-evolves a generator and detector from one base LLM via RLAIF and RLVR, lifting a 7B model to match advanced LLMs on RAGTruth faithfulness detection.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.07964","ref_index":11,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"KronQ: LLM Quantization via Kronecker-Factored Hessian","primary_cat":"cs.LG","submitted_at":"2026-07-08T22:34:52+00:00","verdict":null,"verdict_confidence":null,"novelty_score":null,"formal_verification":null,"one_line_summary":null,"context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.07779","ref_index":57,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"From Solvers to Research: Large Language Model-Driven Formal Mathematics at the Research Frontier","primary_cat":"cs.CL","submitted_at":"2026-07-08T17:46:36+00:00","verdict":"ACCEPT","verdict_confidence":"HIGH","novelty_score":6.0,"formal_verification":"none","one_line_summary":"LLM formal provers must shift from competition solvers to research agents that handle open-ended, under-specified frontier mathematics under machine-checked rigor.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.07674","ref_index":1,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Max Out GRPO Signal: Adaptive Trace Prefix Control for Hard Reasoning Problems","primary_cat":"cs.LG","submitted_at":"2026-07-08T17:32:58+00:00","verdict":"CONDITIONAL","verdict_confidence":"UNKNOWN","novelty_score":6.0,"formal_verification":"none","one_line_summary":"AdaPrefix-GRPO treats solution-prefix length as a feedback controller targeting 50% rollout success rate during GRPO training, then anneals to zero prefix, yielding 1.6–2.1× accuracy gains over vanilla GRPO at matched compute on hard math.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.07646","ref_index":4,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"RL Post-Training Builds Compositional Reasoning Strategies","primary_cat":"cs.AI","submitted_at":"2026-07-08T17:04:42+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":7.0,"formal_verification":"none","one_line_summary":"RL post-training composes primitive rewrite skills into reusable macro and parallel contraction strategies that solve problems inaccessible to the base model under large sampling budgets.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.07554","ref_index":29,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"RubriQ: Rubric-Guided Group Relative Policy Optimization for Constraint-Aware Quantum Circuit Synthesis","primary_cat":"quant-ph","submitted_at":"2026-07-08T15:50:54+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"A rubric-guided GRPO pipeline fine-tunes a 7B LLM to synthesize quantum circuits achieving 3.31x T-gate compression with <1% hardware-constraint violations, validated on IBM and IonQ processors.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.07761","ref_index":94,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Aligning Clinical Needs and AI Capabilities: A Survey on LLMs for Medical Reasoning","primary_cat":"cs.AI","submitted_at":"2026-07-08T15:19:37+00:00","verdict":"ACCEPT","verdict_confidence":"HIGH","novelty_score":6.0,"formal_verification":"none","one_line_summary":"A dual clinical-computational taxonomy for medical LLM reasoning plus a five-level 5k-sample benchmark showing specialists excel at diagnosis and general models at decision support/dialogue.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.07492","ref_index":4,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Search, Fail, Recover: A Training Framework for Correction-Aware Reasoning","primary_cat":"cs.AI","submitted_at":"2026-07-08T14:53:55+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Pyligent trains LLMs to search, detect failures via task validators, and backtrack to recoverable prefixes, improving solve rates by 13–73 points over gold-only SFT on hidden graphs, Sudoku, and Blocksworld.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.07467","ref_index":15,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"SpaCellAgent: A Self-Evolving LLM-Based Multi-Agent Framework for Trajectory Analysis","primary_cat":"cs.AI","submitted_at":"2026-07-08T14:31:46+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":5.0,"formal_verification":"none","one_line_summary":"An LLM multi-agent framework (SpaCellAgent) automates end-to-end trajectory inference on single-cell and spatial transcriptomics data, achieving expert-aligned accuracy with 41.2% faster analysis time.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.07435","ref_index":5,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"RLVP: Penalize the Path, Reward the Outcome","primary_cat":"cs.LG","submitted_at":"2026-07-08T14:06:14+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Pairing outcome rewards with verifiable per-action path penalties reduces constraint violations nearly sixfold at equal task success, while a progress potential accelerates learning only where partial progress is reachable.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.07178","ref_index":8,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Entropy Pacing Policy Optimization for Multi-Task Agentic Reinforcement Learning","primary_cat":"cs.LG","submitted_at":"2026-07-08T09:13:05+00:00","verdict":"ACCEPT","verdict_confidence":"HIGH","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Replacing GRPO's fixed clipping range with a task-wise entropy-aware adaptive bound stabilizes multi-task agentic LLM training by synchronizing exploration-exploitation paces.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.06987","ref_index":10,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"UP: Unbounded Positive Asymmetric Optimization for Breaking the Exploration-Stability Dilemma","primary_cat":"cs.LG","submitted_at":"2026-07-08T04:21:42+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Replacing the importance sampling ratio with a stop-gradient self-anchored ratio for positive advantages yields unclipped, REINFORCE-equivalent gradients that improve exploration without training instability.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.06935","ref_index":102,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Mathematical methods of reinforcement learning","primary_cat":"math.OC","submitted_at":"2026-07-08T02:57:22+00:00","verdict":"ACCEPT","verdict_confidence":"HIGH","novelty_score":0.0,"formal_verification":"none","one_line_summary":"A survey unifying the operator-theoretic, probabilistic, and optimization-based mathematical structures underlying modern reinforcement learning algorithms.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.06875","ref_index":7,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Video2Reaction: Mapping Video to Audience Reaction Distribution in the Wild","primary_cat":"cs.CV","submitted_at":"2026-07-08T00:17:20+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"A new dataset and benchmark maps movie clips to distributions of audience emotional reactions derived from YouTube comments, showing that finetuned vision-language models can predict these distributions from video alone.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.06845","ref_index":60,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"LLMs Silently Correct African American English: Auditing and Mitigating Dialect Bias via Activation Steering","primary_cat":"cs.CL","submitted_at":"2026-07-07T22:36:30+00:00","verdict":"ACCEPT","verdict_confidence":"UNKNOWN","novelty_score":7.0,"formal_verification":"none","one_line_summary":"Six state-of-the-art LLMs systematically prefer Standard American English over AAE continuations, and a training-free activation steering method reduces this bias 5-20x more than prompting while preserving fluency.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.06796","ref_index":16,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Enhancing deep learning models for time series classification via knowledge distillation","primary_cat":"cs.LG","submitted_at":"2026-07-07T20:51:22+00:00","verdict":"CONDITIONAL","verdict_confidence":"HIGH","novelty_score":4.0,"formal_verification":"none","one_line_summary":"Knowledge distillation most benefits intermediate-complexity students for time series classification, cutting parameters sharply while matching teacher accuracy across FCN, Inception, and ConvTran on UCR.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.06720","ref_index":24,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"When Does In-Context Search Help? A Sampling-Complexity Theory of Reflection-Driven Reasoning","primary_cat":"cs.AI","submitted_at":"2026-07-07T18:36:04+00:00","verdict":"CONDITIONAL","verdict_confidence":"HIGH","novelty_score":7.0,"formal_verification":"none","one_line_summary":"When reflections localize early errors, in-context search solves exp-small pass-rate problems with poly sequential attempts; otherwise it offers no asymptotic gain over parallel sampling, and the update is learnable and RLVR-optimal.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.06522","ref_index":14,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Bridging Physical Reasoning and Task Generalization via Visual Action Outcome Reasoning Alignment","primary_cat":"cs.AI","submitted_at":"2026-07-07T17:27:59+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"VAORA aligns VLM chain-of-thought reasoning with visual scene observations and post-action outcomes via structured symbolic rewards, achieving cross-task and cross-environment generalization on physical reasoning benchmarks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.06306","ref_index":28,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"UI2App: Benchmarking Visual Interaction Inference in Executable Web Application Generation","primary_cat":"cs.SE","submitted_at":"2026-07-07T14:08:55+00:00","verdict":"ACCEPT","verdict_confidence":"MODERATE","novelty_score":7.0,"formal_verification":"none","one_line_summary":"UI2App introduces a benchmark showing that vision-language models can reconstruct web page visuals but largely fail to infer the underlying interaction logic from screenshots alone.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.06223","ref_index":19,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Information Gain-based Rollout Policy Optimization: An Adaptive Tree-Structured Rollout Approach for Multi-Turn LLM Agents","primary_cat":"cs.AI","submitted_at":"2026-07-07T12:47:23+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"IGRPO allocates multi-turn LLM agent rollout budget proportional to intermediate-state information gain, inducing an exponentially tilted teacher distribution for policy optimization.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.06175","ref_index":17,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Improving LLM-Generated Process Model Quality Through Reinforcement Learning: The Role of Reward Function Design","primary_cat":"cs.CL","submitted_at":"2026-07-07T11:53:02+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Equal reward weighting outperforms targeted weighting in RL-based BPMN generation across 48 configurations, with design choices producing effects as large as applying RL itself.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.06157","ref_index":108,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"LLM Agents for Deliberative Collaboration: A Study on Joint Decision Making Under Partial Observability","primary_cat":"cs.CL","submitted_at":"2026-07-07T11:34:10+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"A benchmark for LLM agents in partially observable joint decision-making reveals that deliberation challenges current models but can enable reflection and error correction.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.06145","ref_index":13,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Prompting Complexity: Shortest Prompts for Texts and Behaviors in LLMs","primary_cat":"cs.CL","submitted_at":"2026-07-07T11:12:44+00:00","verdict":"CONDITIONAL","verdict_confidence":"HIGH","novelty_score":6.0,"formal_verification":"none","one_line_summary":"The paper defines prompting complexity as the length of the shortest plausible prompt that deterministically generates a target text with a fixed language model.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.06012","ref_index":4,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Structured Data Extraction from Real Estate Documents using Clustering, Classification, and Large Language Models","primary_cat":"cs.CV","submitted_at":"2026-07-07T08:54:11+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":4.0,"formal_verification":"none","one_line_summary":"A pipeline classifies 3965 real-estate questionnaires and extracts 35 structured attributes from 2781 selectable-text documents via DeepSeek R1, reporting Jaccard consistency 0.82.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.05992","ref_index":36,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"PluraMath: Extending Mathematical Reasoning Evaluation Beyond High-Resource Languages","primary_cat":"cs.CL","submitted_at":"2026-07-07T08:25:29+00:00","verdict":"CONDITIONAL","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"PluraMath extends PolyMath with human-validated math problems in 18 mid-to-extreme low-resource languages and benchmarks 27 reasoning LLMs, finding a persistent high- vs low-resource performance gap.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.05911","ref_index":8,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Progressive Reasoning with Primitive Correction for Compositional Zero-Shot Learning","primary_cat":"cs.CV","submitted_at":"2026-07-07T07:06:27+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"PRPC reformulates CZSL as a five-step bidirectional reasoning process in an MLLM with GRPO-based RL post-training, achieving state-of-the-art on three benchmarks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.05863","ref_index":97,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Strategic Bargaining in Multi-Buyer Markets: Reinforcement Learning from Verifiable Rewards for LLM Negotiations","primary_cat":"cs.LG","submitted_at":"2026-07-07T05:41:54+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"RLVR training teaches a 30B LLM to strategically explore a multi-buyer market and extract 70% of available surplus, outperforming frontier models up to 1T parameters in concurrent negotiation.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.05861","ref_index":2,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Mitigating Factual Hallucination in Large Reasoning Models via Mixed-Mode Advantage Regularization","primary_cat":"cs.CL","submitted_at":"2026-07-07T05:34:25+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"MARGO mitigates thinking-induced hallucination in large reasoning models by using mixed-mode GRPO rollout groups that compare thinking trajectories against same-model non-thinking references.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.05716","ref_index":6,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Scene Graph Thinking: Reinforcing Structured Visual Reasoning for Multimodal Large Language Models","primary_cat":"cs.CV","submitted_at":"2026-07-07T01:00:51+00:00","verdict":"ACCEPT","verdict_confidence":"HIGH","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Scene-graph-aligned SFT plus node-as-proxy GRPO rewards let small MLLMs outperform larger baselines on fine-grained visual reasoning tasks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.05391","ref_index":14,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"LLM-as-a-Verifier: A General-Purpose Verification Framework","primary_cat":"cs.AI","submitted_at":"2026-07-06T17:59:35+00:00","verdict":"CONDITIONAL","verdict_confidence":"HIGH","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Expecting over scoring-token logits yields continuous, scalable verification that improves agent trajectory selection and dense RL rewards across coding, robotics, and medical benchmarks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.05378","ref_index":4,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"CompactionRL: Reinforcement Learning with Context Compaction for Long-Horizon Agents","primary_cat":"cs.LG","submitted_at":"2026-07-06T17:55:12+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":5.0,"formal_verification":"none","one_line_summary":"CompactionRL trains LLM agents to generate context summaries during RL rollouts, enabling long-horizon task completion under fixed context budgets with consistent gains on SWE-bench Verified and Terminal-Bench 2.0.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.05184","ref_index":2,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Rethinking On-Policy Self-Distillation for Thinking Models","primary_cat":"cs.AI","submitted_at":"2026-07-06T15:01:35+00:00","verdict":"CONDITIONAL","verdict_confidence":"HIGH","novelty_score":7.0,"formal_verification":"none","one_line_summary":"Privileged-context on-policy self-distillation degrades thinking models' long-budget accuracy by suppressing forking and self-correction behaviors, while helping instruction-tuned models.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.02502","ref_index":2,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"DemoPSD: Disagreement-Modulated Policy Self-Distillation","primary_cat":"cs.LG","submitted_at":"2026-07-02T17:58:29+00:00","verdict":"CONDITIONAL","verdict_confidence":"HIGH","novelty_score":6.5,"formal_verification":"none","one_line_summary":"Disagreement-modulated reverse-KL barycenter targets let on-policy self-distillation attenuate privileged leakage while preserving exploration, beating SDPO and GRPO on SciKnowEval and GPQA.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.02390","ref_index":13,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"DecompRL: Solving Harder Problems by Learning Modular Code Generation","primary_cat":"cs.LG","submitted_at":"2026-07-02T16:25:10+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"DecompRL is an RL method that learns modular code decomposition for LLMs, enabling exponential candidate generation via recombination to solve harder coding problems with lower GPU cost.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.02374","ref_index":14,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"DRIFTLENS: Measuring Memory-Induced Reasoning Drift in Personalized Language Models","primary_cat":"cs.AI","submitted_at":"2026-07-02T16:15:25+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"User-attribute memory induces measurable medium-to-large reasoning drift in LLMs above pragmatic noise, only partly reduced by GRPO/DPO post-training.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.02291","ref_index":8,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Optimizing Visual Generative Models via Distribution-wise Rewards","primary_cat":"cs.LG","submitted_at":"2026-07-02T15:08:56+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Distribution-wise rewards with subset-replace strategy and post-hoc merging improve FID-50K on SiT (8.30 to 5.77) and EDM2 (3.74 to 3.52) while preserving diversity.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.02234","ref_index":5,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Purified OPSD: On-Policy Self-Distillation Without Losing How to Think","primary_cat":"cs.AI","submitted_at":"2026-07-02T14:33:07+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"Purified OPSD subtracts a reference-only teacher's signal from standard OPSD supervision and applies PMI to create a cleaner distillation target, yielding gains on long-CoT models while preserving epistemic behavior.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.02073","ref_index":1,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Evidence-State Rewards for Long-Context Reasoning","primary_cat":"cs.AI","submitted_at":"2026-07-02T12:11:49+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"Maven is an RL method using answer-conditioned evidence-state values to assign rewards to add, link, and drop actions on evidence memory, outperforming outcome-only baselines on LongBench v2, LongReason, and RULER.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.02047","ref_index":8,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"OpenSafeIntent: Evaluating Intent-Calibrated Safe Completion Across Dual-Use Prompt Sets","primary_cat":"cs.CL","submitted_at":"2026-07-02T11:14:52+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"OpenSafeIntent benchmark shows models fail to calibrate safety across intent shifts in matched dual-use prompts, indicating current evaluations are insufficient.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.01883","ref_index":23,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"PairCoder++: Pair Programming as a Universal Paradigm for Verified Code-Driven Multimodal and Structured-Artifact Generation","primary_cat":"cs.CL","submitted_at":"2026-07-02T08:36:02+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"PairCoder is a two-agent pair-programming method that leverages toolchain verification oracles to improve LLM generation of verifiable structured artifacts on 17 benchmarks across seven models.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.01855","ref_index":20,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Regression Accumulation in Multi-Turn LLM Programming Conversations","primary_cat":"cs.SE","submitted_at":"2026-07-02T08:15:40+00:00","verdict":"CONDITIONAL","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"Regression accumulation affects 40-73% of 8-turn LLM coding tasks on extended HumanEval+/MBPP+ benchmarks, with verification gates improving final-turn pass rates on prior tests.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.01647","ref_index":26,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"AgenticDataBench: A Comprehensive Benchmark for Data Agents","primary_cat":"cs.DB","submitted_at":"2026-07-02T03:18:59+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"AgenticDataBench is a new benchmark covering realistic data science tasks across 15 domains using extracted skills and LLM-generated workflows to evaluate data agents at fine granularity.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.01571","ref_index":44,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Geometric Signatures of Reasoning: A Spectral Perspective on Task Hardness","primary_cat":"cs.LG","submitted_at":"2026-07-02T01:03:49+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Introduces effective dimension d_ρ from spectral analysis of reasoning trajectories to distinguish task hardness (0.93 AUC on MATH500) and uses kinematic features for early correctness prediction from partial generations.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.01480","ref_index":6,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Procedural Memory Distillation: Online Reflection for Self-Improving Language Models","primary_cat":"cs.AI","submitted_at":"2026-07-01T21:20:57+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"PMD extracts and distills cross-episode procedural knowledge from RL rollouts into LLM policies at three abstraction levels, yielding 3.8-13.6% gains over SDPO on SCIKNOWEVAL and LIVECODEBENCH via co-evolution.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.01470","ref_index":5,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"World Feedback for Clinical Agents: Diagnosing RL in FHIR Environments","primary_cat":"cs.AI","submitted_at":"2026-07-01T21:02:54+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"MedAgentBench-v3 shows capability ceilings and format-knowledge barriers limit pure RL to 18.2% while rule-based SFT reaches 34.1% on clinical protocol tasks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.01465","ref_index":3,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Beyond Next-Token Prediction: An RLVR Proof of Concept for Tool-Use Agents on Atlassian Workflows","primary_cat":"cs.AI","submitted_at":"2026-07-01T20:55:07+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":4.0,"formal_verification":"none","one_line_summary":"RLVR training on five synthetic Atlassian API environments raises average tool-use reward for Qwen models from 0.35-0.92 to 0.95-1.00 on four non-degenerate scenarios.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.01233","ref_index":49,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Measuring the Gap Between Human and LLM Research Ideas","primary_cat":"cs.CL","submitted_at":"2026-07-01T17:59:56+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"LLM-generated research ideas cluster more around bridge-like opportunities and synthesis methods than the broader distribution seen in human papers.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.01191","ref_index":37,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Perceive-to-Reason: Decoupling Perception and Reasoning for Fine-Grained Visual Reasoning","primary_cat":"cs.CV","submitted_at":"2026-07-01T17:24:26+00:00","verdict":"UNVERDICTED","verdict_confidence":"MODERATE","novelty_score":7.0,"formal_verification":"none","one_line_summary":"P2R decouples perception from reasoning in VLMs via a two-stage process and PRA-GRPO alternating RL training, reporting gains such as 93.2% on V-Star for the 4B model over its Qwen3-VL backbone.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.01170","ref_index":4,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Diffusion-GR2: Diffusion Generative Reasoning Re-ranker","primary_cat":"cs.IR","submitted_at":"2026-07-01T17:02:20+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"CFT + on-policy distillation + RL converts an AR reasoning re-ranker into a block-diffusion model that recovers near-AR accuracy at 2.4–3.5× decode throughput on Amazon Beauty.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.01120","ref_index":14,"ref_count":2,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents","primary_cat":"cs.DC","submitted_at":"2026-07-01T16:08:02+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":4.0,"formal_verification":"none","one_line_summary":"Current agentic RL systems lack three key components needed for self-evolving agents at scale, requiring new co-designed architectures such as AReaL2.0 to enable policy updates from deployed workloads.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.01083","ref_index":7,"ref_count":2,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Staleness-Learning Rate Scaling Laws for Asynchronous RLHF","primary_cat":"cs.LG","submitted_at":"2026-07-01T15:40:12+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Stale rollouts introduce O(S * eta) surrogate-gradient bias in async GRPO, yielding stability condition eta << min{R_batch / (S * G_upd), R_crit / (T * G_upd)} under smoothness assumptions.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.01061","ref_index":30,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Agentic generation of verifiable rules for deterministic, self-expanding reaction classification","primary_cat":"cs.AI","submitted_at":"2026-07-01T15:24:06+00:00","verdict":"CONDITIONAL","verdict_confidence":"HIGH","novelty_score":7.0,"formal_verification":"none","one_line_summary":"Multi-agent LLMs classify USPTO reactions and write verified SMIRKS rules, expanding a reaction taxonomy from 68 to 14,073 classes and matching proprietary classifiers on held-out and out-of-distribution data.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.01050","ref_index":10,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"GeoSearcher: Anchor-Guided Progressive Reasoning for Remote Sensing Visual Grounding with Process Supervision","primary_cat":"cs.CV","submitted_at":"2026-07-01T15:12:51+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"GeoSearcher introduces anchor-centric reasoning supervised fine-tuning and process-faithful group relative policy optimization to improve MLLM-based remote sensing visual grounding.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.00862","ref_index":2,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"CAT: Confidence-Adaptive Thinking for Efficient Reasoning of Large Reasoning Models","primary_cat":"cs.CL","submitted_at":"2026-07-01T12:27:14+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"CAT uses intrinsic confidence signals in preference optimization to adapt reasoning length in LRMs, outperforming uniform compression baselines on accuracy across benchmarks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.00711","ref_index":11,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"ClarifyCodeBench: Evaluating LLMs on Clarifying Ambiguous Requirements for Code Generation","primary_cat":"cs.SE","submitted_at":"2026-07-01T09:58:59+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"ClarifyCodeBench is a new benchmark with manual annotations and two metrics showing that LLMs strong at code generation are weak at clarifying ambiguous requirements, with performance worsening as ambiguity density rises.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.00604","ref_index":30,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Vehicle Routing Problem Meets Large Language Models: An Overview and Perspectives","primary_cat":"math.OC","submitted_at":"2026-07-01T08:30:20+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":4.0,"formal_verification":"none","one_line_summary":"Survey organizing LLM uses for VRP into modeler, designer, and coordinator roles, covering variants, solvers, benchmarks, and two experiments.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.00531","ref_index":19,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Active-GRPO: Adaptive Imitation and Self-Improving Reasoning for Molecular Optimization","primary_cat":"cs.LG","submitted_at":"2026-07-01T07:22:46+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Active-GRPO reaches 0.1773 average SRxSim on TOMG-Bench MOLOPT by adaptively switching between imitation and self-reinforcement while upgrading references, outperforming GRPO and RePO.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.00511","ref_index":32,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Large Language Models for Multi-Lingual Equivalent Mutant Detection: An Extended Empirical Study","primary_cat":"cs.SE","submitted_at":"2026-07-01T06:46:13+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"LLM-based methods achieve higher F1-scores than traditional approaches for equivalent mutant detection in Java and C, with fine-tuned code embeddings performing best and showing cross-lingual generalization.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.00482","ref_index":53,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Know When to Stop: Segment-Level Credit Assignment for Reducing Overthinking","primary_cat":"cs.CL","submitted_at":"2026-07-01T06:09:56+00:00","verdict":null,"verdict_confidence":null,"novelty_score":null,"formal_verification":null,"one_line_summary":null,"context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.00422","ref_index":24,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"KidnapRAG: A Black-Box Attack for Hijacking Reasoning in Agentic Retrieval-Augmented Generation Systems","primary_cat":"cs.CR","submitted_at":"2026-07-01T04:32:17+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"KidnapRAG is a sequential black-box poisoning attack on Agentic RAG systems using Bait, Chain-Link, and Mal-Ins documents to redirect retrieval and reasoning, outperforming prior baselines.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.00341","ref_index":11,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"DiscoLoop: Looping Discrete Embeddings and Continuous Hidden States for Multi-hop Reasoning","primary_cat":"cs.CL","submitted_at":"2026-07-01T02:32:02+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":7.0,"formal_verification":"none","one_line_summary":"DiscoLoop adds a decoded token-embedding channel to looped transformers, fixing a representation mismatch that limited implicit multi-hop reasoning and improving OOD generalization.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.00164","ref_index":2,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Verifiable Rewards for Calibrated Probabilistic Forecasting","primary_cat":"cs.LG","submitted_at":"2026-06-30T20:42:45+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"A verifiable empirical win rate reward combined with gradient masking enables RL training of a 7B model to reach betting-market calibration on NFL win probabilities using only outcome data.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.00152","ref_index":2,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"GRPO, Dr. GRPO, and DAPO Are Three Operations on One Number: The Group-Standard-Deviation Identity","primary_cat":"cs.LG","submitted_at":"2026-06-30T20:28:08+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"GRPO, Dr. GRPO, and DAPO are three settings of one dial on the group standard deviation of binary rewards, unified by the group-standard-deviation identity where disagreement equals update magnitude.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.00115","ref_index":11,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"PixelEyes: Decoupling Perception and Reasoning for Pinpoint Visual Evidence Seeking","primary_cat":"cs.CV","submitted_at":"2026-06-30T19:51:54+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"PixelEyes decouples reasoning and perception via mask-guided search and semantic BFS, introduces PixelEyes-6K dataset and Pinpoint-Bench benchmark, and open-sources code and models.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.32032","ref_index":37,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Reinforcement Learning with Metacognitive Feedback Elicits Faithful Uncertainty Expression in LLMs","primary_cat":"cs.CL","submitted_at":"2026-06-30T17:56:01+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"RLMF uses quality of model self-judgments to refine RL rankings and select training data, achieving SOTA faithful calibration while preserving accuracy and outperforming standard RL by up to 63%.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.32029","ref_index":2,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"When LLMs Read Tables Carelessly: Measuring and Reducing Data Referencing Errors","primary_cat":"cs.CL","submitted_at":"2026-06-30T17:54:50+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"LLMs exhibit data referencing errors across model sizes; a critic model detects them at 78.2% F1 and boosts accuracy up to 12% via filtering and rejection sampling.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.31986","ref_index":15,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"CoLT: Teaching Multi-Modal Models to Think with Chain of Latent Thoughts","primary_cat":"cs.CV","submitted_at":"2026-06-30T17:24:40+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"CoLT replaces text chain-of-thought in multimodal LLMs with three supervised latent vectors, improving average accuracy from 75.7 to 79.1 across eight benchmarks while cutting text-decoding time 22.6x.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.31748","ref_index":11,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Addressing Over-Refusal in LLMs with Competing Rewards","primary_cat":"cs.LG","submitted_at":"2026-06-30T14:38:49+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"SEAR trains one LLM via adversarial process rewards to explore harmful reasoning paths but flip to safe outputs, reducing over-refusal while preserving safety.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.31693","ref_index":63,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"ShopX: A Foundation Model for Intent-to-Item Fulfillment in Agentic Shopping","primary_cat":"cs.IR","submitted_at":"2026-06-30T14:05:28+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"A single LLM trained to emit semantic item codes can fulfill complex shopping intents with fewer tool hand-offs, improving multi-turn follow-up on Taobao-derived tasks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.31650","ref_index":4,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"ECHO: Prune To Act, Trace To Learn With Selective Turn Memory In Agentic RL","primary_cat":"cs.LG","submitted_at":"2026-06-30T13:29:58+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"ECHO stores each agent turn as a source-indexed memory, reconstructs bounded contexts by selecting useful records, and routes RL credit through the same selection trace — reaching 43.4% on BrowseComp-Plus vs 28.9% (GRPO) and 36.1% (SUPO).","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.31608","ref_index":175,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"CLExEval: A Human-in-the-Loop Framework for Qualitative Evaluation of LLM Clinical Reasoning","primary_cat":"cs.CL","submitted_at":"2026-06-30T12:56:42+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"CLExEval introduces a human-annotated evaluation framework on 40 rare cases that identifies verbosity bias, hidden knowledge paradox, and 68.6% reasoning-to-output mismatch in LLMs while showing LLM-as-a-Judge overestimates reliability.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.31585","ref_index":39,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"DPPE: Rethinking Camera-Based Positional Encoding for Scaling Multi-View Transformers","primary_cat":"cs.CV","submitted_at":"2026-06-30T12:38:58+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"DPPE decouples rotation and translation in camera positional encodings for multi-view transformers to resolve late-stage training stagnation and improve generalization in novel view synthesis.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.31580","ref_index":10,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"LASER: Load-Aware Serving with Early-Exit for Reasoning LLMs at the Edge","primary_cat":"cs.DC","submitted_at":"2026-06-30T12:35:48+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"LASER reduces edge LLM serving latency by 17-38% and improves SLO satisfaction by 3-6% via load-aware adaptive early-exit thresholds and difficulty-aware budget pre-allocation, with 1% average accuracy cost.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.31392","ref_index":6,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"ReGRPO: Reflection-Augmented Policy Optimization for Tool-Using Agents","primary_cat":"cs.AI","submitted_at":"2026-06-30T09:19:38+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"ReGRPO augments group-relative policy optimization with a reflective data engine that generates ErrorType-Evidence-FixPlan triplets from near-miss tool actions to improve recovery in multimodal agents.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.31307","ref_index":5,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"When the Database Fails: Prompting LLM Dialogue Agents for Safe Recovery in Task-Oriented Dialogue","primary_cat":"cs.CL","submitted_at":"2026-06-30T08:18:11+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Guided-Retry prompting cuts hallucination from 30.5% to 15.3% on MultiWOZ and 20.9% to 12.2% on SGD in LLM dialogue agents facing database failures.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.31048","ref_index":3,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Knowledge Distillation from Large Reasoning Models to Compact Student Models: A Case Study on the John O Bryan Mathematics Competition","primary_cat":"cs.LG","submitted_at":"2026-06-30T02:34:45+00:00","verdict":"CONDITIONAL","verdict_confidence":"LOW","novelty_score":4.0,"formal_verification":"none","one_line_summary":"Distilling CoT from DeepSeek-R1 to Qwen2.5-7B on competition problems yields 4.76 pp accuracy gain to 69.43% and 73.1% on MATH-500, with accuracy falling as response length decreases.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.31039","ref_index":21,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Truth or Sophistry? LoFa: A Benchmark for LLM Robustness Against Logical Fallacies","primary_cat":"cs.CL","submitted_at":"2026-06-30T02:17:45+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"LoFa is a new benchmark and LFR@k metric for measuring LLM resistance to sustained logical fallacy attacks via generated question-argument pairs and debate simulations.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.30989","ref_index":26,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Wait, am I Being Fair? Characterizing Deductive Stereotyping and Mitigating It with Fair-GCG","primary_cat":"cs.CL","submitted_at":"2026-06-30T00:00:42+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"The paper characterizes deductive stereotyping in LLMs and introduces Fair-GCG to discover injection phrases that improve fairness across benchmarks, reasoning, and real-world tasks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.30923","ref_index":23,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Behavior Cloning is Not All You Need: The Optimality of On-Policy Distillation for Noisy Expert Feedback","primary_cat":"cs.LG","submitted_at":"2026-06-29T21:18:21+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":8.0,"formal_verification":"none","one_line_summary":"Noisy expert imitation learning requires exponential samples for offline methods but polynomial for a variant of on-policy distillation under a noise condition.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.30562","ref_index":24,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Morphing into Hybrid Attention Models","primary_cat":"cs.CL","submitted_at":"2026-06-29T17:02:34+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"FlashMorph formulates hybrid layer selection as budget-constrained optimization, trains per-layer gates on synthetic retrieval data with linearization regularization, then discretizes and distills to produce efficient hybrid architectures.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.30556","ref_index":19,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Poller: Are LLMs Suitable for Evaluating the Poetry Understanding Task?","primary_cat":"cs.CL","submitted_at":"2026-06-29T16:51:31+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Poller reduces LLM-human disagreement in evaluating Chinese poetry understanding by having LLMs role-play as authors, with reported error reductions of 94.55% and 89.53% on rhetorical techniques and defamiliarization.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.30445","ref_index":9,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"When Does Online Imitation Learning Help in LLM Post-Training? The Role of (Non-)Realizability Beyond Horizon","primary_cat":"cs.LG","submitted_at":"2026-06-29T15:17:42+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"Online IL overcomes an information-theoretic bottleneck that offline IL faces in non-realizable settings even at horizon 1, under a new structural characterization of reward-relative misspecification.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.30442","ref_index":16,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"The FIL Hypothesis: Inductive Biases Help with Kernel Engineering","primary_cat":"cs.AI","submitted_at":"2026-06-29T15:16:33+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"The FIL Hypothesis claims that inductive biases outperform purely data-driven methods on GPU programming tasks with non-trivial feedback loops.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.30251","ref_index":63,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"TACO: Tool-Augmented Credit Optimization for Agentic Tool Use","primary_cat":"cs.MA","submitted_at":"2026-06-29T13:01:05+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"TACO combines Differential Answer-Probe Reward (DAPR) and Outcome-Gated Advantage Routing (OGAR) to assign credit to tool calls in agentic visual reasoning, producing accuracy gains on multimodal benchmarks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.30217","ref_index":18,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Before Thinking, Learn to Decide: Proactive Routing for Efficient Visual Reasoning","primary_cat":"cs.CL","submitted_at":"2026-06-29T12:30:24+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"PRP introduces proactive routing via Draft Rating Learning and Joint Rating Learning to route queries early between draft and target models for efficient multimodal reasoning.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.30175","ref_index":51,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"CORTEX: High-Quality Cross-Domain Organization of Web-Scale Corpora through Ontological Corpus Graph","primary_cat":"cs.CL","submitted_at":"2026-06-29T11:51:00+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"Cortex uses an Ontological Corpus Graph to structure web-scale corpora, creating a refined 24.14B-token corpus and a new benchmark validated on eight LLMs.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.30077","ref_index":2,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Online Data Selection for Instruction Tuning via Gaussian Processes","primary_cat":"cs.LG","submitted_at":"2026-06-29T10:08:12+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"GAIA models continuous utility with Gaussian processes across semantic space and applies fixed-share Hedge updates to achieve dynamic regret guarantees while outperforming baselines on three datasets.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.29982","ref_index":12,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Beyond Uniform Experts: Cost-Aware Expert Execution for Efficient Multi-Device MoE Inference","primary_cat":"cs.DC","submitted_at":"2026-06-29T08:57:59+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"CAEE reduces MoE inference latency 8-18% on 671B DeepSeek-R1 by cost-aware expert pruning and low-overhead compensation while keeping accuracy drop under 1%.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.29823","ref_index":16,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Experience Graphs: The Data Foundation for Self-Improving Agents","primary_cat":"cs.DB","submitted_at":"2026-06-29T06:02:20+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Trellis treats agent experience graphs as first-class database state so that search patterns become queries, enabling crash recovery, scaling, and closed-loop training as architectural byproducts.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.29812","ref_index":10,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Consistency as Inductive Bias: Learning Cross-View Invariance for Robust Multimodal Reasoning","primary_cat":"cs.CV","submitted_at":"2026-06-29T05:45:18+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":4.0,"formal_verification":"none","one_line_summary":"ConsistRoll enforces cross-view consistency during RLVR training for MLLMs by joint rewards on grouped original and augmented views, yielding robustness gains on math, general, and hallucination benchmarks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.29799","ref_index":21,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"The CRISTAL Method: Neurosymbolic analysis from AI-synthesized world models","primary_cat":"cs.AI","submitted_at":"2026-06-29T05:27:09+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"CRISTAL is a neurosymbolic framework that synthesizes interpretable probabilistic world models from language priors for full Bayesian analysis and budget-aware data acquisition, claiming Bayes-optimal accuracy on synthetic equity classification with 5 examples.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.29773","ref_index":12,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"GLIP: Graph and LLM Joint Pretraining for Graph-Level Tasks","primary_cat":"cs.LG","submitted_at":"2026-06-29T04:30:45+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"GLIP is a joint GNN-LLM pretraining framework that uses augmentation, multi-token selection, a diffusion projector, and combined contrastive plus semantic losses to boost graph classification and reasoning after fine-tuning on limited labels.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.29758","ref_index":33,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"PS-PPO: Prefix-Sampling PPO for Critic-Free RLHF","primary_cat":"cs.LG","submitted_at":"2026-06-29T04:04:32+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":7.0,"formal_verification":"none","one_line_summary":"PS-PPO samples a per-trajectory cutoff and importance-weights truncated gradients, preserving the full critic-free update in expectation while cutting RLHF training compute and memory.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null}],"limit":100,"offset":0}}