{"total":15,"items":[{"citing_arxiv_id":"2607.07001","ref_index":54,"ref_count":1,"confidence":0.55,"is_internal_anchor":false,"paper_title":"Ego-Human Motion Prediction with 3D-Aware LLM","primary_cat":"cs.CV","submitted_at":"2026-07-08T04:51:24+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Ego3DLM jointly predicts past and future 3D body pose and motion descriptions in a single autoregressive pass, conditioned on egocentric video, 3D scene features, and three-point tracking, achieving state-of-the-art on the Nymeria benchmark.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.06909","ref_index":35,"ref_count":1,"confidence":0.55,"is_internal_anchor":false,"paper_title":"Seeing What Matters: Lesion-Aware High-Resolution Patch Discovery and Fusion for Chest X-ray Report Generation","primary_cat":"cs.CV","submitted_at":"2026-07-08T02:02:03+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":7.0,"formal_verification":"none","one_line_summary":"LePaX enables high-resolution chest X-ray report generation by learning to allocate resolution to diagnostically relevant regions and fusing high-res patches back into global features without increasing token count.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.01754","ref_index":41,"ref_count":1,"confidence":0.55,"is_internal_anchor":false,"paper_title":"Path-level Hindsight Instructions for Semantic Exploration in Vision-Language Navigation","primary_cat":"cs.AI","submitted_at":"2026-07-02T06:11:07+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Phi-Nav generates path-level hindsight instructions from on-policy exploration trajectories to supply additional semantic supervision for vision-language navigation agents.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.31292","ref_index":16,"ref_count":1,"confidence":0.55,"is_internal_anchor":false,"paper_title":"AtomiMed: Hierarchical Atomic Fact-Checking for Universal Clinical-Aware Medical Report Evaluation","primary_cat":"cs.CE","submitted_at":"2026-06-30T08:07:45+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"AtomiMed is a new modality-agnostic evaluation framework for medical report generation that decomposes reports into hierarchical atomic clinical facts and applies agentic cross-verification to achieve higher correlation with radiologist judgments than n-gram metrics.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.31099","ref_index":29,"ref_count":1,"confidence":0.55,"is_internal_anchor":false,"paper_title":"Seeing Through Multiple Views: Parameter-Efficient Fine-Tuning via Selective Neurons for Consistent Radiology Report Generation","primary_cat":"cs.CV","submitted_at":"2026-06-30T03:48:20+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"View-PNDF detects and selectively fine-tunes view-specific neurons for consistent multi-view chest X-ray report generation, followed by LLM consolidation of reports.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.28393","ref_index":16,"ref_count":1,"confidence":0.55,"is_internal_anchor":false,"paper_title":"Transition-Aware best-of-N sampling for Longitudinal Chest X-ray Reports","primary_cat":"cs.CV","submitted_at":"2026-06-23T23:11:59+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"Transition-aware best-of-N sampling embeds report sentences as sets, computes directional transition vectors via set-to-set distances, and scores candidates by proximity to ground-truth training transitions.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.19027","ref_index":15,"ref_count":3,"confidence":0.55,"is_internal_anchor":false,"paper_title":"MedFM-Robust: Benchmarking Robustness of Medical Foundation Models","primary_cat":"cs.CV","submitted_at":"2026-05-18T18:50:56+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"A new robustness benchmark for medical VLMs and segmentation models shows fine-tuning strategy dominates performance under 40 perturbation types, with medical-specific ones hitting segmentation hardest.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.23969","ref_index":19,"ref_count":1,"confidence":0.55,"is_internal_anchor":false,"paper_title":"SLAP: Stratified Loss-based Pruning for On-Policy Data-Efficient Instruction Tuning","primary_cat":"cs.CL","submitted_at":"2026-05-13T06:36:24+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"SLAP is a new batch-aware pruning framework that uses distribution-aware stratified sampling and Hessian-approximated gradients to select data, claiming 20-40% less data while matching or exceeding full-dataset performance on LLM instruction tuning tasks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.11208","ref_index":17,"ref_count":2,"confidence":0.55,"is_internal_anchor":false,"paper_title":"Hi-GaTA: Hierarchical Gated Temporal Aggregation Adapter for Surgical Video Report Generation","primary_cat":"cs.CV","submitted_at":"2026-05-11T20:21:34+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Hi-GaTA is a hierarchical gated temporal aggregation adapter that uses short-to-long temporal pyramids and gated fusion to enable surgical video report generation, backed by a new 214-video benchmark and a surgical ViViT pretrained on 40,000 minutes of video.","context_count":1,"top_context_role":"background","top_context_polarity":"background","context_text":"5B/3B [26]) while keeping the video encoder(Sur40k) fixed. In addi- tion, we compare against two strong MLLM baselines, LLaVA-Med-v1.5 [8] and Qwen2.5-VL-7B [2] specifically prompted to follow the target reporting format. Following standard practice for report generation, we evaluate with a five- dimensionalsuitecapturingbothlexicaloverlapandsemanticsimilarity:BLEU[17], ROUGE-L [11], METEOR [3], MedBERTScore [5], and CIDEr [23]. All metrics are computed between generated reports and reference reports. Results Analysis.Our quantitative results are summarized in Table 1 and Table 2. Table 1 demonstrates the critical role of surgical-domain perception in processing long-horizon videos. Specifically, replacingSur40kwith general-"},{"citing_arxiv_id":"2605.08787","ref_index":14,"ref_count":2,"confidence":0.55,"is_internal_anchor":false,"paper_title":"Lost in Volume: The CT-SpatialVQA Benchmark for Evaluating Semantic-Spatial Understanding of 3D Medical Vision-Language Models","primary_cat":"cs.CV","submitted_at":"2026-05-09T08:16:00+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"CT-SpatialVQA benchmark reveals that eight 3D medical VLMs achieve only 34% average accuracy on semantic-spatial reasoning tasks from CT data, frequently below random performance.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.23309","ref_index":27,"ref_count":1,"confidence":0.55,"is_internal_anchor":false,"paper_title":"STAND: Semantic Anchoring Constraint with Dual-Granularity Disambiguation for Remote Sensing Image Change Captioning","primary_cat":"cs.CV","submitted_at":"2026-04-25T13:50:11+00:00","verdict":null,"verdict_confidence":null,"novelty_score":null,"formal_verification":null,"one_line_summary":null,"context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.21926","ref_index":65,"ref_count":1,"confidence":0.55,"is_internal_anchor":false,"paper_title":"Seeing Without Eyes: 4D Human-Scene Understanding from Wearable IMUs","primary_cat":"cs.CV","submitted_at":"2026-04-23T17:59:16+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"IMU-to-4D uses wearable IMU data and repurposed LLMs to predict coherent 4D human motion plus coarse scene structure, outperforming cascaded state-of-the-art pipelines in temporal stability.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.10385","ref_index":39,"ref_count":1,"confidence":0.55,"is_internal_anchor":false,"paper_title":"GTASA: Ground Truth Annotations for Spatiotemporal Analysis, Evaluation and Training of Video Models","primary_cat":"cs.CV","submitted_at":"2026-04-12T00:01:51+00:00","verdict":"CONDITIONAL","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"GEST-Engine turns game engines into zero-cost dense ground-truth video generators; GTASA reveals frozen video encoders fail inter-entity spatial relation probes.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.05341","ref_index":16,"ref_count":1,"confidence":0.55,"is_internal_anchor":false,"paper_title":"Curr-RLCER:Curriculum Reinforcement Learning For Coherence Explainable Recommendation","primary_cat":"cs.IR","submitted_at":"2026-04-07T02:25:36+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":4.0,"formal_verification":"none","one_line_summary":"Curr-RLCER applies curriculum reinforcement learning with coherence-driven rewards to align generated explanations with predicted ratings in explainable recommendation systems.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2512.24366","ref_index":23,"ref_count":1,"confidence":0.55,"is_internal_anchor":false,"paper_title":"On the Factual Consistency of Text-based Explainable Recommendation Models","primary_cat":"cs.IR","submitted_at":"2025-12-30T17:25:15+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"A prompting pipeline and statement-level metrics show that six state-of-the-art text-based explainable recommendation models achieve high semantic similarity but very low factual consistency on Amazon review data.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null}],"limit":50,"offset":0}