{"total":15,"items":[{"citing_arxiv_id":"2606.30988","ref_index":6,"ref_count":2,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Multisensory Continual Learning: Adapting Pretrained Visuomotor Policies to Force","primary_cat":"cs.RO","submitted_at":"2026-06-29T23:58:07+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"MuSe adapts a vision-action policy to force-torque via multi-stage fusion, multisensory future prediction, and experience replay, improving both new contact-rich tasks and original vision tasks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.29941","ref_index":73,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Seeing Touch from Motion: A Unified Modality-Aware Visuo-Tactile Policy with Tactile Motion Correlation","primary_cat":"cs.RO","submitted_at":"2026-06-29T08:20:18+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"A visuo-tactile policy learning method that exploits tactile motion correlation for contact state distinction and Mixture-of-Transformers for cross-modal fusion.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.12069","ref_index":188,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Tac-DINO: Learning Vision-Tactile Features with Patch Alignment","primary_cat":"cs.CV","submitted_at":"2026-06-10T13:33:42+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Tac-DINO constructs a large tactile dataset and Vis-Tac Holographic Matching Benchmark, then proposes Vision-Tactile Patch Alignment (VTPA) methods that outperform non-aligned baselines on local-to-global feature matching.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.08765","ref_index":22,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"RGB-S: Image-Aligned Tactile Saliency for Robust Dexterous Manipulation","primary_cat":"cs.RO","submitted_at":"2026-06-07T18:07:01+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"RGB-S projects tactile contacts onto images as force-modulated Gaussian saliency maps via kinematics and zero-initialized conditioning, raising real-world occluded dexterous manipulation success by 26.7 percentage points over implicit baselines.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.08737","ref_index":66,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Dream-Tac: A Unified Tactile World Action Model for Contact-Rich Robot Manipulation","primary_cat":"cs.RO","submitted_at":"2026-06-07T17:18:23+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Dream-Tac unifies visual and tactile signals in a world action model using contact-gated fusion and attention bias, reporting 31.7% average action accuracy gains on six manipulation tasks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.03392","ref_index":35,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"OpenEAI-Platform: An Open-source Embodied Artificial Intelligence Hardware-Software Unified Platform","primary_cat":"cs.RO","submitted_at":"2026-06-02T09:34:08+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"OpenEAI-Platform delivers an open-source low-cost robotic arm and VLA model that outperforms commercial arms and matches large pretrained baselines on four real-world manipulation tasks using limited open data.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.20894","ref_index":32,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Mobile UMI: Cross-View Diffusion Policy with Decoupled Kinematics for Mobile Manipulation","primary_cat":"cs.RO","submitted_at":"2026-05-20T08:33:53+00:00","verdict":"CONDITIONAL","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"A hardware-free dual-camera capture framework with ChArUco spatial unification and receding-horizon state alignment enables decoupled SE(3) manipulation and SE(2) base trajectories for diffusion policies, yielding 83.8% average success on four long-horizon household tasks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.17336","ref_index":49,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Tactile-based Multimodal Fusion in Embodied Intelligence: A Survey of Vision, Language, and Contact-Driven Paradigms","primary_cat":"cs.RO","submitted_at":"2026-05-17T09:09:30+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":4.0,"formal_verification":"none","one_line_summary":"A survey proposing a hierarchical taxonomy for multimodal tactile fusion datasets and methods across perception, generation, and interaction in embodied intelligence.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.07308","ref_index":48,"ref_count":2,"confidence":0.9,"is_internal_anchor":false,"paper_title":"AT-VLA: Adaptive Tactile Injection for Enhanced Feedback Reaction in Vision-Language-Action Models","primary_cat":"cs.RO","submitted_at":"2026-05-08T06:17:08+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"AT-VLA proposes adaptive tactile injection and a dual-stream tactile reaction mechanism to enhance VLA models for contact-rich robotic manipulation with real-time responses.","context_count":1,"top_context_role":"background","top_context_polarity":"background","context_text":"Vtla: Vision-tactile-language- action model with preference learning for insertion manipula- tion.arXiv preprint arXiv:2505.09577, 2025. [47] Zongzheng Zhang, Haobo Xu, Zhuo Yang, Chenghao Yue, Zehao Lin, Huan-ang Gao, Ziwei Wang, and Hao Zhao. Ta- vla: Elucidating the design space of torque-aware vision- language-action models.arXiv preprint arXiv:2509.07962, 2025. [48] Xinyue Zhu, Binghao Huang, and Yunzhu Li. Touch in the wild: Learning fine-grained manipulation with a portable visuo-tactile gripper.arXiv preprint arXiv:2507.15062, 2025. [49] Brianna Zitkovich, Tianhe Yu, Sichun Xu, Peng Xu, Ted Xiao, Fei Xia, Jialin Wu, Paul Wohlhart, Stefan Welker, Ayzaan Wahid, et al. Rt-2: Vision-language-action models transfer"},{"citing_arxiv_id":"2604.27224","ref_index":8,"ref_count":2,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Learning Tactile-Aware Quadrupedal Loco-Manipulation Policies","primary_cat":"cs.RO","submitted_at":"2026-04-29T21:46:58+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"A hierarchical tactile-aware policy trained from human demos and sim RL improves real quadrupedal loco-manipulation by 28.54% on average over vision-only and visuotactile baselines.","context_count":1,"top_context_role":"background","top_context_polarity":"background","context_text":"II. RELATEDWORKS To enable low-cost and portable demonstration collection in diverse real-world settings, recent work has adopted UMI-style handheld gripper interfaces [6]. This paradigm supports in-the- wild data collection and enables hardware-agnostic diffusion policies that transfer across robot embodiments [7]. Building on the same interface, [8] further incorporates tactile arrays into the gripper to capture vision and touch simultaneously. A growing body of work has shown that tactile feedback is particularly valuable for contact-rich manipulation, support- ing tasks such as power-plug insertion [9]-[11], light-bulb screwing [12], cable untangling [13], and peeling [3]. For quadrupedal robots, [14] presents a system that mounts tactile"},{"citing_arxiv_id":"2604.21331","ref_index":81,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"FingerViP: Learning Real-World Dexterous Manipulation with Fingertip Visual Perception","primary_cat":"cs.RO","submitted_at":"2026-04-23T06:37:22+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"FingerViP equips each finger with a miniature camera and trains a multi-view diffusion policy that achieves 80.8% success on real-world dexterous tasks previously limited by wrist-camera occlusion.","context_count":1,"top_context_role":"background","top_context_polarity":"background","context_text":"as regression-based method (e.g., MSE [ 7]), discretization- based method [ 55, 38], and cluster-based methods (e.g., K- means [ 23, 8]), have limitations when modeling complex action distributions and struggle to effectively capture the diversity and nuances of human behavior [ 50]. In addition, some alternative IL paradigms have been proposed, including IRL [ 81], GAIL [ 26], and DAgger [ 55]. However, these IL paradigms typically involve high interaction costs, computa- tional complexity, and limited scalability in real-world robotic systems [32, 47]. Over the past few years, Denoising Diffusion Probabilistic Models (DDPMs) [ 27] have emerged as a powerful generative framework and have been increasingly adopted in robotics for"},{"citing_arxiv_id":"2604.13015","ref_index":44,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Learning Versatile Humanoid Manipulation with Touch Dreaming","primary_cat":"cs.RO","submitted_at":"2026-04-14T17:54:17+00:00","verdict":null,"verdict_confidence":null,"novelty_score":null,"formal_verification":null,"one_line_summary":null,"context_count":1,"top_context_role":"background","top_context_polarity":"background","context_text":"Systems (RSS), 2025. [42] B. Huang, Y . Wang, X. Yang, Y . Luo, and Y . Li, \"3d-vitac: Learning fine-grained manipulation with visuo-tactile sensing,\"arXiv preprint arXiv:2410.24091, 2024. [43] X. Zhu, B. Huang, and Y . Li, \"Touch in the wild: Learning fine-grained manipulation with a portable visuo-tactile gripper,\"arXiv preprint arXiv:2507.15062, 2025. [44] H. Chen, J. Xu, H. Chen, K. Hong, B. Huang, C. Liu, J. Mao, Y . Li, Y . Du, and K. Driggs-Campbell, \"Multi-modal manipulation via multi- modal policy consensus,\"arXiv preprint arXiv:2509.23468, 2025. [45] T. Lin, Y . Zhang, Q. Li, H. Qi, B. Yi, S. Levine, and J. Malik, \"Learning visuotactile skills with two multifingered hands,\" in2025 IEEE International Conference on Robotics and Automation (ICRA)."},{"citing_arxiv_id":"2603.03243","ref_index":50,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"HoMMI: Learning Whole-Body Mobile Manipulation from Human Demonstrations","primary_cat":"cs.RO","submitted_at":"2026-03-03T18:36:49+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"HoMMI learns whole-body mobile manipulation policies from robot-free human demonstrations by augmenting UMI with egocentric sensing and bridging the embodiment gap through an agnostic visual representation, relaxed head actions, and a whole-body controller.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2601.20239","ref_index":79,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"TouchGuide: Inference-Time Steering of Visuomotor Policies via Touch Guidance","primary_cat":"cs.RO","submitted_at":"2026-01-28T04:22:47+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"TouchGuide improves contact-rich robot manipulation by steering diffusion or flow-matching visuomotor policies with tactile feasibility scores from a contrastively trained Contact Physical Model.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2509.23468","ref_index":39,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Multi-Modal Manipulation via Multi-Modal Policy Consensus","primary_cat":"cs.RO","submitted_at":"2025-09-27T19:43:04+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"A policy that factorizes into modality-specific diffusion models combined by a learned router network for adaptive multi-modal robotic manipulation.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null}],"limit":50,"offset":0}