{"total":29,"items":[{"citing_arxiv_id":"2605.27976","ref_index":31,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"VoiceGiraffe: A Benchmark for Extreme Long-Context Audio-Language Understanding","primary_cat":"cs.SD","submitted_at":"2026-05-27T05:15:14+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"VoiceGiraffe is a new benchmark showing that long-range memory persistence remains a key bottleneck for large audio language models on hour-scale audio.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.24202","ref_index":4,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"When Does Multi-Agent RL Improve LLM Workflows? Workflow, Scale, and Policy-Sharing Tradeoffs","primary_cat":"cs.AI","submitted_at":"2026-05-22T20:43:30+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Multi-agent RL on LLM workflows improves base models depending on workflow, task, and scale, with Isolated-Policy reaching higher peaks but more terminal accuracy cliffs than Shared-Policy.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.25572","ref_index":3,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Dictionary learning for Kernel EDMD","primary_cat":"math.DS","submitted_at":"2026-04-28T12:41:20+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"A dictionary learning method optimizes weighted kernels via gradients for kEDMD to approximate Koopman operators, with pruning of unimportant kernels based on learned weights.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.24623","ref_index":33,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"XGRAG: A Graph-Native Framework for Explaining KG-based Retrieval-Augmented Generation","primary_cat":"cs.AI","submitted_at":"2026-04-27T15:52:20+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"XGRAG uses graph perturbations to quantify component contributions in GraphRAG and achieves 14.81% better explanation quality than text-based baselines on QA datasets, with correlations to graph centrality.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.23933","ref_index":4,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"KT4EQG: Personalized Exercise Question Generation via Knowledge Tracing","primary_cat":"cs.CY","submitted_at":"2026-04-24T02:14:57+00:00","verdict":"CONDITIONAL","verdict_confidence":"UNKNOWN","novelty_score":5.0,"formal_verification":"none","one_line_summary":"KT4EQG combines a knowledge tracing model for concept selection with an alignment-trained LLM for question generation, achieving higher simulated exam scores than baseline generators on two educational datasets.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.15482","ref_index":4,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Harmonizing Multi-Objective LLM Unlearning via Unified Domain Representation and Bidirectional Logit Distillation","primary_cat":"cs.LG","submitted_at":"2026-04-16T19:09:17+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"A multi-objective LLM unlearning approach standardizes data into unified domain representations and applies bidirectional logit distillation to align objectives and achieve balanced state-of-the-art results across efficacy, utility, boundary preservation, and robustness.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.14568","ref_index":4,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Learning Adaptive Reasoning Paths for Efficient Visual Reasoning","primary_cat":"cs.CV","submitted_at":"2026-04-16T02:59:30+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"AVR trains vision-language models to adaptively select among full reasoning, perception-only, or direct-answer formats using a modified policy optimization method, reducing token use by 50-90% with little accuracy loss.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.13993","ref_index":4,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Reward Design for Physical Reasoning in Vision-Language Models","primary_cat":"cs.AI","submitted_at":"2026-04-15T15:36:26+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Accuracy-based rewards outperform SFT and other reward variants in GRPO training of VLMs on the PhyX physics benchmark, with attention-weight rewards raising spatial reasoning accuracy from 0.27 to 0.50.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.13991","ref_index":4,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Adaptive Conformal Prediction for Improving Factuality of Generations by Large Language Models","primary_cat":"cs.CL","submitted_at":"2026-04-15T15:35:42+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"An adaptive conformal prediction approach for LLMs enables prompt-dependent calibration that improves conditional coverage for factuality while preserving marginal guarantees and supporting selective prediction.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.13883","ref_index":4,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Context Sensitivity Improves Human-Machine Visual Alignment","primary_cat":"cs.CV","submitted_at":"2026-04-15T13:47:08+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Context-sensitive similarity computation from embeddings improves odd-one-out accuracy by up to 15% over context-insensitive baselines for human visual alignment.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.14251","ref_index":4,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Calibrate-Then-Delegate: Safety Monitoring with Risk and Budget Guarantees via Model Cascades","primary_cat":"cs.LG","submitted_at":"2026-04-15T11:05:59+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"CTD trains a lightweight DV probe to predict escalation benefits and calibrates its threshold via multiple hypothesis testing on held-out data to deliver finite-sample guarantees on delegation rate while outperforming uncertainty-based cascades on safety tasks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.11416","ref_index":4,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Exact Certification of Neural Networks and Partition Aggregation Ensembles against Label Poisoning","primary_cat":"cs.LG","submitted_at":"2026-04-13T13:01:14+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":8.0,"formal_verification":"none","one_line_summary":"EnsembleCert and ScaLabelCert enable tighter and exact certificates for neural network robustness against label-flipping attacks by leveraging white-box information and neural tangent kernel equivalence.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.11201","ref_index":4,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"CocoaBench: Evaluating Unified Digital Agents in the Wild","primary_cat":"cs.CL","submitted_at":"2026-04-13T09:00:10+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"CocoaBench shows the best tested unified digital agents succeed on only 45.1% of human-designed tasks that demand integrated vision, search, and coding.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.11035","ref_index":4,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Introspective Diffusion Language Models","primary_cat":"cs.AI","submitted_at":"2026-04-13T06:01:01+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"I-DLM matches same-scale autoregressive model quality in diffusion language models by enforcing introspective consistency via strided decoding, outperforming prior DLMs on 15 benchmarks including 69.6 on AIME-24.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.09408","ref_index":42,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"HiL-Bench (Human-in-Loop Benchmark): Do Agents Know When to Ask for Help?","primary_cat":"cs.AI","submitted_at":"2026-04-10T15:21:44+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"HiL-Bench shows frontier AI agents fail to ask for help on incomplete tasks, recovering only a fraction of full-information performance, but RL training on Ask-F1 reward improves judgment and transfers across domains.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.09406","ref_index":4,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"OASIS: Online Activation Subspace Learning for Memory-Efficient Training","primary_cat":"cs.LG","submitted_at":"2026-04-10T15:19:54+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"OASIS tracks an evolving low-dimensional activation subspace to project activations, gradients, and optimizer states, cutting peak memory up to 2x versus full fine-tuning while matching performance on finetuning and pretraining tasks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.09181","ref_index":40,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"MixFlow: Mixed Source Distributions Improve Rectified Flows","primary_cat":"cs.CV","submitted_at":"2026-04-10T10:06:15+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Mixing unconditional Gaussian noise with a κ-conditioned source during training of rectified flows reduces path curvature, yielding 12% better FID scores and faster sampling than standard rectified flows.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.08425","ref_index":4,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Learning Who Disagrees: Demographic Importance Weighting for Modeling Annotator Distributions with DiADEM","primary_cat":"cs.AI","submitted_at":"2026-04-09T16:29:17+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"DiADEM learns demographic importance weights to model annotator disagreement distributions and outperforms LLM and neural baselines on disagreement tracking in DICES and VOICED benchmarks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.02022","ref_index":43,"ref_count":2,"confidence":0.9,"is_internal_anchor":false,"paper_title":"ATBench: A Diverse and Realistic Agent Trajectory Benchmark for Safety Evaluation and Diagnosis","primary_cat":"cs.AI","submitted_at":"2026-04-02T13:26:20+00:00","verdict":null,"verdict_confidence":null,"novelty_score":null,"formal_verification":null,"one_line_summary":null,"context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.01306","ref_index":4,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"M2-Verify: A Large-Scale Multidomain Benchmark for Checking Multimodal Claim Consistency","primary_cat":"cs.CL","submitted_at":"2026-04-01T18:18:10+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"M2-Verify is a new multidomain benchmark dataset for multimodal scientific claim consistency that reveals state-of-the-art models drop from 85.8% to 61.6% Micro-F1 on complex perturbations and produce hallucinated explanations.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2512.09538","ref_index":4,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Don't Throw Away Your Beams: Improving Consistency-based Uncertainties in LLMs via Beam Search","primary_cat":"stat.ML","submitted_at":"2025-12-10T11:24:29+00:00","verdict":"CONDITIONAL","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Beam search for candidate generation in consistency-based UQ for LLMs reduces variance and improves performance over multinomial sampling on six QA datasets, supported by a theoretical lower bound on beam-set probability mass.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2511.18203","ref_index":72,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"SkillWrapper: Generative Predicate Invention for Task-level Robot Planning","primary_cat":"cs.RO","submitted_at":"2025-11-22T22:25:11+00:00","verdict":"REJECT","verdict_confidence":"HIGH","novelty_score":6.0,"formal_verification":"none","one_line_summary":"SkillWrapper learns human-readable skill models from images via VLM predicate invention, but its provable soundness/completeness claim is not established.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2511.06424","ref_index":4,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Turbo-DDCM: Fast and Flexible Zero-Shot Diffusion-Based Image Compression","primary_cat":"eess.IV","submitted_at":"2025-11-09T15:41:27+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Turbo-DDCM accelerates DDCM-based zero-shot image compression by batching noise vectors per step while preserving performance and adding priority-aware and PSNR-targeted variants.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2509.26383","ref_index":49,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Efficient and Transferable Agentic Knowledge Graph RAG via Reinforcement Learning","primary_cat":"cs.CL","submitted_at":"2025-09-30T15:14:24+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"KG-R1 trains a single RL agent to retrieve from and reason over knowledge graphs in one loop, achieving higher accuracy with fewer tokens than multi-module baselines and transferring to unseen graphs.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2509.25758","ref_index":60,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Thinking Sparks!: Emergent Attention Heads in Reasoning Models During Post Training","primary_cat":"cs.AI","submitted_at":"2025-09-30T04:23:43+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Post-training on reasoning tasks sparks the emergence of specialized attention heads that enable structured computation, with SFT adding stable heads while GRPO uses dynamic activation and pruning tied to reward signals, and controllable think models relying on compensatory heads instead of specific","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2509.05489","ref_index":59,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Self-Aligned Reward: Towards Effective and Efficient Reasoners","primary_cat":"cs.LG","submitted_at":"2025-09-05T20:39:43+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Self-aligned reward uses relative perplexity differences to encourage concise, query-specific reasoning in LLMs, yielding 4% accuracy gains and 30% lower inference cost when added to PPO or GRPO.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2502.19731","ref_index":46,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Preference Learning Unlocks LLMs' Psycho-Counseling Skills","primary_cat":"cs.CL","submitted_at":"2025-02-27T03:50:25+00:00","verdict":"CONDITIONAL","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"A new expert-principle preference dataset enables an 8B LLM to reach 87% win rate vs GPT-4o on counseling responses through standard preference optimization.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2412.02125","ref_index":62,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Preference Goal Tuning: Post-Training as Latent Control for Frozen Policies","primary_cat":"cs.AI","submitted_at":"2024-12-03T03:27:48+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"PGT optimizes latent goal embeddings for frozen policies via trajectory-level preference objectives, reporting 72-81.6% relative gains on 17 Minecraft tasks and 13.4% better OOD performance than fine-tuning.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2408.16286","ref_index":86,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Near-Optimal Policy Identification in Robust Constrained Markov Decision Processes via Epigraph Form","primary_cat":"cs.LG","submitted_at":"2024-08-29T06:37:16+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"Presents the first algorithm to identify an ε-optimal policy in robust constrained MDPs via epigraph form and bisection search with Õ(ε^{-4}) robust policy evaluations.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null}],"limit":50,"offset":0}