{"total":18,"items":[{"citing_arxiv_id":"2607.07029","ref_index":12,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Gimitest: A Comprehensive Tool for Testing Reinforcement Learning Policies","primary_cat":"cs.LG","submitted_at":"2026-07-08T06:00:40+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":4.0,"formal_verification":"none","one_line_summary":"Gimitest is an open-source tool that decorates RL environment APIs to enable search-based, metamorphic, and adversarial testing of single- and multi-agent policies.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.29867","ref_index":3,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"RoAd-RL: A Unified Library and Benchmark for Robust Adversarial Reinforcement Learning","primary_cat":"cs.LG","submitted_at":"2026-06-29T07:03:45+00:00","verdict":"CONDITIONAL","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"RoAd-RL is a new benchmarking library for adversarial reinforcement learning that evaluates DQN, PPO, and SAC agents across 192 attack-defense configurations and finds substantial robustness variations plus cases where defenses harm performance more than attacks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.29012","ref_index":104,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"The Game Changer Problem: Controlling Equilibria with Discrete Rewards","primary_cat":"cs.GT","submitted_at":"2026-06-27T17:17:54+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Introduces the game changer problem and supplies feasibility characterizations plus dynamic programming algorithms for forcing a target equilibrium under discrete reward constraints in two-player games.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.10371","ref_index":39,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Test-time Adversarial Takeover: A Real-time Hijacking Interface against Robotic Diffusion Policies","primary_cat":"cs.RO","submitted_at":"2026-06-09T03:31:09+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":8.0,"formal_verification":"none","one_line_summary":"TAKO demonstrates real-time adversarial takeover of robotic diffusion policies via reusable universal patches on visual inputs, achieving 100% success in steering attacker-chosen trajectories across multiple tasks, encoders, and diffusion methods.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.04314","ref_index":41,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Testing Neural Networks via Bayesian-Guided Exploration of Decision Landscapes","primary_cat":"cs.LG","submitted_at":"2026-06-03T00:42:10+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"BayesWarp is a testing framework combining saliency techniques and uncertainty-aware Bayesian optimization to discover diverse model failures on image classification tasks while preserving data proximity.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.04310","ref_index":45,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Latent Anchor-Driven Test Generation for Deep Neural Networks","primary_cat":"cs.LG","submitted_at":"2026-06-03T00:33:21+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Latte performs seed-centered one-step latent mutations along class anchors in VQ-VAE space to produce diverse, low-drift, fault-revealing DNN tests.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.03724","ref_index":11,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Same Weights, Different Robot: A Deployment Safety View of VLA Policies","primary_cat":"cs.CR","submitted_at":"2026-06-02T14:45:00+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"The paper identifies a deployment safety gap in VLA policies where identical checkpoints can be executable-inequivalent due to action metadata mismatches, supported by a derived closed-form transform and empirical drift measurements on LIBERO benchmarks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.18058","ref_index":23,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Threats to Arabic Handwriting Recognition: Investigating Black-Box Adversarial Attacks on embedded ConvNet models","primary_cat":"cs.CV","submitted_at":"2026-05-18T08:45:16+00:00","verdict":"CONDITIONAL","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Black-box attacks, especially Pixle, reach 99-100% success on Arabic handwriting ConvNet models across two benchmark datasets while preserving character structure.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.12792","ref_index":42,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"SoK: A Comprehensive Analysis of the Current Status of Neural Tangent Generalization Attacks with Research Directions","primary_cat":"cs.LG","submitted_at":"2026-05-12T22:10:01+00:00","verdict":"ACCEPT","verdict_confidence":"LOW","novelty_score":3.0,"formal_verification":"none","one_line_summary":"NTGA is the first clean-label generalization attack under black-box settings but is vulnerable to adversarial training and image transformations, with newer attacks outperforming it.","context_count":1,"top_context_role":"background","top_context_polarity":"background","context_text":"Generalization attacks are also known asavailability attacks [104], and their main goal is to degrade overall model accuracy, including validation and test accuracy. On the other hand, integrity attacks cause the model to misclassify on specific images in a clean test set and degrade the test accuracy of the model.Poison Frogs[74] is one of the major integrity attacks [42][105]. Unlike generalization attacks, integrity attacks do not mitigate validation accuracy. Adversarial attacks: While data poisoning attacks harm the model at training time, adversarial attacks occur at test time. Attackers add imperceptible perturbation to the test data so that pre-trained models misclassify them with high confidentiality during test time"},{"citing_arxiv_id":"2605.16312","ref_index":2,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"When Actions Disappear: Adversarial Action Removal in Self-Play Reinforcement Learning","primary_cat":"cs.LG","submitted_at":"2026-05-04T14:05:39+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Adversarial action removal in self-play RL inflicts greater damage than random masking or learned perturbations, persists across algorithms and domains, transfers between agents, and resists recovery through extended training.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.02495","ref_index":124,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Efficient Preference Poisoning Attack on Offline RLHF","primary_cat":"cs.LG","submitted_at":"2026-05-04T11:45:38+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"Preference poisoning against log-linear DPO reduces to a binary sparse approximation problem solved by lattice-reduction (BAL-A) and matching-pursuit (BMP-A) algorithms that carry recovery guarantees.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2603.28281","ref_index":5,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Corruption-robust Offline Multi-agent Reinforcement Learning From Human Feedback","primary_cat":"cs.LG","submitted_at":"2026-03-30T11:03:36+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"Introduces robust estimators for linear Markov games in offline MARLHF that achieve O(ε^{1-o(1)}) or O(√ε) bounds on Nash or CCE gaps under uniform or unilateral coverage.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2512.09893","ref_index":39,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"A Speculative GLRT-Backed Approach for Robust Deep Learning-Based Array Processing","primary_cat":"eess.SP","submitted_at":"2025-12-10T18:19:44+00:00","verdict":null,"verdict_confidence":null,"novelty_score":null,"formal_verification":null,"one_line_summary":null,"context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2510.01479","ref_index":7,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Density-Ratio Weighted Behavioral Cloning: Learning Control Policies from Corrupted Datasets","primary_cat":"cs.LG","submitted_at":"2025-10-01T21:43:04+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Weighted BC estimates trajectory density ratios from a clean reference set via binary discrimination and reweights the BC loss to converge to the clean expert policy with finite-sample bounds independent of contamination rate.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2502.03698","ref_index":28,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"How Vulnerable Is My Learned Policy? Universal Adversarial Perturbation Attacks On Modern Behavior Cloning Policies","primary_cat":"cs.LG","submitted_at":"2025-02-06T01:17:39+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"Modern imitation learning methods including Diffusion Policy and Implicit Behavior Cloning are highly vulnerable to universal adversarial perturbations, with successful black-box transfer attacks across algorithms.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2502.02844","ref_index":6,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Wolfpack Adversarial Attack for Robust Multi-Agent Reinforcement Learning","primary_cat":"cs.LG","submitted_at":"2025-02-05T02:59:23+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Wolfpack attack framework disrupts MARL cooperation by targeting initial and assisting agents; WALL trains robust policies against it with reported experimental gains.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2406.09250","ref_index":33,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"MirrorCheck: Efficient Adversarial Defense for Vision-Language Models","primary_cat":"cs.CV","submitted_at":"2024-06-13T15:55:04+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"MirrorCheck detects adversarial attacks on VLMs via T2I regeneration for semantic consistency checks, using stochastic model selection and one-time perturbations for robustness against adaptive attacks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"1906.12061","ref_index":10,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Learning to Cope with Adversarial Attacks","primary_cat":"cs.LG","submitted_at":"2019-06-28T07:10:30+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"MLAH agent in deep RL demonstrates hierarchical coping mechanisms and improved reward maintenance under spaced adversarial attacks, at the expense of stability.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null}],"limit":50,"offset":0}