{"total":20,"items":[{"citing_arxiv_id":"2607.05391","ref_index":86,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"LLM-as-a-Verifier: A General-Purpose Verification Framework","primary_cat":"cs.AI","submitted_at":"2026-07-06T17:59:35+00:00","verdict":"CONDITIONAL","verdict_confidence":"HIGH","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Expecting over scoring-token logits yields continuous, scalable verification that improves agent trajectory selection and dense RL rewards across coding, robotics, and medical benchmarks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.31958","ref_index":23,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Adapting Generalist Robot Policies with Semantic Reinforcement Learning","primary_cat":"cs.RO","submitted_at":"2026-06-30T17:00:33+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"SARL optimizes language prompt inputs to generalist vision-language-action policies through online RL to solve complex long-horizon tasks by composing existing skills.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.29892","ref_index":30,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Trust Your Instincts: Confidence-Driven Test-Time RL for Vision-Language-Action Models","primary_cat":"cs.RO","submitted_at":"2026-06-29T07:31:41+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"T^2VLA is a test-time reinforcement learning framework for VLAs that uses internal confidence to define intrinsic rewards via similarity to high-confidence expert demonstrations and a dual-expert bootstrapping mechanism.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.26588","ref_index":10,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Inference-Time Robot Behavior Steering through Physically-Aware Reconfiguration of Task-Structure","primary_cat":"cs.RO","submitted_at":"2026-06-25T04:20:08+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"ReStruct steers robot policies at inference time by reconfiguring task structure with neural automata and synchronous products, claiming up to 25% gains over VLA models in success and preference adherence.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.23640","ref_index":61,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Learning Process Rewards via Success Visitation Matching for Efficient RL","primary_cat":"cs.LG","submitted_at":"2026-06-22T17:30:24+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Success Visitation Matching uses a discriminator to turn sparse outcome rewards into dense process rewards by matching visitations of successful episodes, provably preserving the optimal policy and speeding up robotic RL finetuning.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.22860","ref_index":6,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"HiL-ResRL: A Model-Agnostic Finetuning Adapter via Human-in-the-loop Residual Reinforcement Learning","primary_cat":"cs.RO","submitted_at":"2026-06-22T05:07:08+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"HiL-ResRL trains a model-agnostic residual policy on VLA actions using human-guided online RL, achieving over 95% success rate after 1.5 hours of real-robot training.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.21572","ref_index":26,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Robot Critics that Sweat the Small Stuff","primary_cat":"cs.RO","submitted_at":"2026-06-19T16:14:12+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Fine-tuning VLMs with pairwise progress supervision from policy rollouts improves fine-grained failure detection and boosts robot manipulation success by 11% real-world and 5.9% in simulation.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.19656","ref_index":10,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"DF-ExpEnse: Diffusion Filtered Exploration for Sample Efficient Finetuning","primary_cat":"cs.RO","submitted_at":"2026-06-17T23:40:45+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"DF-ExpEnse improves sample efficiency in finetuning diffusion-based robotic policies by filtering diffusion-generated actions with critic ensembles and enabling fleet-level collaboration.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.11087","ref_index":45,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Test-Time Gradient Guidance of Flow Policies in Reinforcement Learning","primary_cat":"cs.LG","submitted_at":"2026-06-09T16:45:57+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"QGF performs test-time policy optimization for flow models in RL by guiding a behavior-cloned reference policy with value-function gradients, achieving strong results on high-dimensional offline RL benchmarks without additional policy training.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.10568","ref_index":33,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"VeriSpace: Spatially Grounded Action Verification for Vision-Language-Action Models","primary_cat":"cs.RO","submitted_at":"2026-06-09T08:31:59+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"VeriSpace is a 3D-aware action verifier that improves test-time action selection in VLA models by encoding scenes with visual and geometric information and reasoning over spatial relations and goal progress.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.30660","ref_index":13,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"BOKBO (Best of K Bad Options): Calibrated Abstention for VLA Policies","primary_cat":"cs.LG","submitted_at":"2026-05-28T23:39:09+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"BOKBO is the first conformal abstention method for K-sample VLA policies that supplies finite-sample distribution-free guarantees on executed violation rates, with global and Mondrian per-task variants.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.11479","ref_index":19,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Offline Policy Evaluation for Manipulation Policies via Discounted Liveness Formulation","primary_cat":"cs.RO","submitted_at":"2026-05-12T03:54:30+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":7.0,"formal_verification":"none","one_line_summary":"A liveness-based Bellman operator enables conservative offline policy evaluation for manipulation tasks by encoding task progression and reducing truncation bias from finite horizons.","context_count":1,"top_context_role":"background","top_context_polarity":"background","context_text":"This shows that our method remains useful in low-data regimes. F . Ablation on Embedding Space While the default image encoder used in all experiments is SigLIP2 [27], we additionally conduct an ablation on the choice of image encoder in the cloth folding task with the full dataset (150 episodes). We run our method with SigLIP2 [27], DINOv2 [18], and CLIP [19], with quantitative results across all metrics shown in Fig. 6. Our method performs comparably across the three encoders on all metrics, indicating that it is largely insensitive to the choice of image encoder. This is expected, as large pretrained image encoders are generally competent at retaining task- relevant features. So long as the necessary information is"},{"citing_arxiv_id":"2605.01194","ref_index":15,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"VLA-ATTC: Adaptive Test-Time Compute for VLA Models with Relative Action Critic Model","primary_cat":"cs.RO","submitted_at":"2026-05-02T02:13:11+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"VLA-ATTC equips VLA models with adaptive test-time compute via an uncertainty clutch and relative action critic, cutting failure rates by over 50% on LIBERO-LONG.","context_count":1,"top_context_role":"method","top_context_polarity":"use_method","context_text":"contextual information by attending to raw features of the corresponding layerlin the VLM backbone,F l vlm: Ol raw =Attention(Q(X l−1), K(F l vlm), V(F l vlm))(14) 3.Query Cross-Attention:The third branch attends to the specialized, distilled context features from the VLM's learnable query tokens from the same layerl,F l query: Ol query =Attention(Q(X l−1), K(F l query), V(F l query)) (15) These three outputs are then fused. The query cross- attention branch is modulated by a learnable scalar gat- ing parameter gl, which allows the model to dynamically control the influence of the distilled context. The fused representationO l f used is created by concatenation: Ol f used =Concat[O l sa, Ol raw, gl ×O l query](16) This fused vector is then passed through the standard Trans-"},{"citing_arxiv_id":"2604.24661","ref_index":42,"ref_count":2,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Agent-Centric Observation Adaptation for Robust Visual Control under Dynamic Perturbations","primary_cat":"cs.RO","submitted_at":"2026-04-27T16:24:40+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"ACO-MoE recovers 95.3% of clean-input performance in visual control tasks under Markov-switching corruptions by routing restoration experts and anchoring representations to clean foreground masks.","context_count":1,"top_context_role":"background","top_context_polarity":"background","context_text":"DMControl Generalization with random-color and video-background perturbations, demonstrating a high level of robustness. *: These authors contributed equally 1. Introduction Visual reinforcement learning (RL) enables agents to learn optimal policies directly from visual observations, achieving remarkable success across a wide range of control tasks, from simulated benchmarks [42, 66, 13, 41, 76, 51, 78, 21] to robotic manipulation and autonomous navigation [84, 82, 67, 24, 80, 62, 54, 22, 20]. In contrast to proactive LLM agents that focus on high-level cognitive assistance [58, 23], our work targets low-level RL agents robust to raw visual disturbances. However, deploying learned policies in real-world environments exposes a critical vulnerability:visual perturbations that corrupt the agent's observations can"},{"citing_arxiv_id":"2604.23121","ref_index":57,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Breaking Lock-In: Preserving Steerability under Low-Data VLA Post-Training","primary_cat":"cs.RO","submitted_at":"2026-04-25T03:18:07+00:00","verdict":"UNVERDICTED","verdict_confidence":"UNKNOWN","novelty_score":6.0,"formal_verification":"none","one_line_summary":"DeLock mitigates lock-in in low-data VLA post-training via visual grounding preservation and test-time contrastive prompt guidance, outperforming baselines across eight evaluations while matching data-heavy generalist policies.","context_count":1,"top_context_role":"background","top_context_polarity":"background","context_text":"[55] S. Huang, Q. Chen, X. Zhang, J. Sun, and M. Schwager. Particleformer: A 3d point cloud world model for multi-object, multi-material robotic manipulation.arXiv preprint arXiv:2506.23126, 2025. [56] F. Koulischer, J. Deleu, G. Raya, T. Demeester, and L. Ambrogioni. Dynamic negative guid- ance of diffusion models.arXiv preprint arXiv:2410.14398, 2024. [57] M. Nakamoto, O. Mees, A. Kumar, and S. Levine. Steering your generalists: Improving robotic foundation models via value guidance.arXiv preprint arXiv:2410.13816, 2024. [58] A. Wagenmaker, M. Nakamoto, Y . Zhang, S. Park, W. Yagoub, A. Nagabandi, A. Gupta, and S. Levine. Steering your diffusion policy with latent space reinforcement learning.arXiv"},{"citing_arxiv_id":"2604.19730","ref_index":15,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"FASTER: Value-Guided Sampling for Fast RL","primary_cat":"cs.LG","submitted_at":"2026-04-21T17:52:17+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"FASTER models multi-candidate denoising as an MDP and trains a value function to filter actions early, delivering the performance of full sampling at lower cost in diffusion RL policies.","context_count":1,"top_context_role":"background","top_context_polarity":"background","context_text":"class for a wide variety of domains because they can represent rich, multimodal data distributions [5, 6, 9, 10, 11, 12, 13, 14]. Several recent works study how to improve these policies with value functions or reduce their inference cost. Value-Guided Policy Steering (V-GPS) re-ranks actions from frozen robot policies with a learned value function [15], RoboMonkey scales test-time sampling 2 and verification for vision-language-action policies [ 16], and EXPO couples an expressive base policy with a lightweight edit policy and selects value-maximizing actions online [5]. Other work amortizes or changes the policy itself: One-Step Diffusion Policy distills multi-step denoising into a faster actor [17], while DSRL post-trains a behavior-cloned diffusion policy by running RL in its"},{"citing_arxiv_id":"2604.08508","ref_index":33,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Sumo: Dynamic and Generalizable Whole-Body Loco-Manipulation","primary_cat":"cs.RO","submitted_at":"2026-04-09T17:49:40+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Test-time steering of pre-trained whole-body policies via sample-based planning lets legged robots generalize dynamic loco-manipulation to varied heavy objects and tasks without additional training or tuning.","context_count":1,"top_context_role":"background","top_context_polarity":"background","context_text":"[31] Andrew Y Ng, Daishi Harada, and Stuart Russell. Policy invariance under reward transformations: Theory and application to reward shaping. InProceedings of the Sixteenth International Conference on Machine Learning (ICML), pages 278-287, 1999. [32] Chaoyi Pan, Zeji Yi, Guanya Shi, and Guannan Qu. Model-based diffusion for trajectory optimization. 2024. URL https://arxiv.org/abs/2407.01573. [33] Haozhi Qi, Brent Yi, Mike Lambeta, Yi Ma, Roberto Calandra, and Jitendra Malik. From simple to complex skills: The case of in-hand object reorientation.arXiv preprint arXiv:2501.05439, 2025. [34] Junbin Qiu, Zhengpeng Xie, Xiangda Yan, Yongjie Yang, and Yao Shu. Zeroth-order optimization is se- cretly single-step policy optimization.arXiv preprint"},{"citing_arxiv_id":"2604.06168","ref_index":42,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Action Images: End-to-End Policy Learning via Multiview Video Generation","primary_cat":"cs.CV","submitted_at":"2026-04-07T17:59:30+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"Action Images turn robot arm motions into interpretable multiview pixel videos, letting video backbones serve as zero-shot policies for end-to-end robot learning.","context_count":1,"top_context_role":"background","top_context_polarity":"background","context_text":"arXiv preprint arXiv:2210.02747 (2022) [40] Liu, Z., Li, S., Cousineau, E., Feng, S., Burchfiel, B., Song, S.: Geometry-aware 4d video generation for robot manipulation. arXiv preprint arXiv:2507.01099 (2025) [41] Ljung,L.,Glad,T.:Modelingofdynamicsystems.Prentice-Hall,Inc.(1994) Action Images: End-to-End Policy Learning via Multiview Video Generation 23 [42] Nakamoto, M., Mees, O., Kumar, A., Levine, S.: Steering your generalists: Improving robotic foundation models via value guidance. arXiv preprint arXiv:2410.13816 (2024) [43] Pan, Z., Yang, Z., Zhu, X., Zhang, L.: Efficient4d: Fast dynamic 3d object generation from a single-view video. arXiv preprint arXiv:2401.08742 (2024) [44] Pearce, T., Rashid, T."},{"citing_arxiv_id":"2603.15757","ref_index":24,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"You've Got a Golden Ticket: Improving Generative Robot Policies With A Single Noise Vector","primary_cat":"cs.RO","submitted_at":"2026-03-16T18:00:19+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"A single constant initial noise vector found by Monte-Carlo search improves frozen diffusion and flow-matching robot policies on 46 of 51 tasks, by up to 55% absolute success rate.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2506.15799","ref_index":54,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Steering Your Diffusion Policy with Latent Space Reinforcement Learning","primary_cat":"cs.RO","submitted_at":"2025-06-18T18:35:57+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"DSRL steers pretrained diffusion policies for robotics by applying RL to their latent noise inputs, achieving sample-efficient real-world adaptation with only black-box access.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null}],"limit":50,"offset":0}