{"total":29,"items":[{"citing_arxiv_id":"2607.00293","ref_index":33,"ref_count":1,"confidence":0.55,"is_internal_anchor":false,"paper_title":"Rosetta: Composable Native Multimodal Pretraining","primary_cat":"cs.CV","submitted_at":"2026-07-01T00:42:40+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Rosetta proposes a composable multimodal pretraining method with MAOP to prevent catastrophic forgetting when expanding modalities beyond standard MoE and MoT approaches.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.25046","ref_index":14,"ref_count":1,"confidence":0.55,"is_internal_anchor":false,"paper_title":"TinyFormer: Preserving Tiny Objects in YOLO-DETR Hybrid Real-time Detectors","primary_cat":"cs.CV","submitted_at":"2026-05-24T12:42:13+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"TinyFormer adds Parallel Bi-fusion Module and Spatial Semantic Adapter to a YOLO-DETR hybrid, raising small-object AP by 1.6 points to 58.5% on MS COCO while keeping real-time speed.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.22168","ref_index":40,"ref_count":1,"confidence":0.55,"is_internal_anchor":false,"paper_title":"Measuring Cross-Modal Synergy: A Benchmark for VLM Explainability","primary_cat":"cs.AI","submitted_at":"2026-05-21T08:39:46+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"Introduces Synergistic Faithfulness metric based on Shapley Interaction Index to evaluate cross-modal synergy in VLM explainers, revealing over-reliance on visual salience in existing methods.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.21541","ref_index":25,"ref_count":1,"confidence":0.55,"is_internal_anchor":false,"paper_title":"Frequency-Domain Regularized Adversarial Alignment for Transferable Attacks against Closed-Source MLLMs","primary_cat":"cs.CR","submitted_at":"2026-05-20T08:15:56+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"FRA-Attack uses high-pass DCT feature alignment and frequency-domain gradient regularization to boost adversarial transferability across 15 MLLMs from 7 vendors.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.20600","ref_index":22,"ref_count":1,"confidence":0.55,"is_internal_anchor":false,"paper_title":"Head-Aware Key-Value Compression for Efficient Autoregressive Image Generation","primary_cat":"cs.CV","submitted_at":"2026-05-20T01:30:33+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":7.0,"formal_verification":"none","one_line_summary":"HeadKV compresses KV cache for autoregressive image generation via head-aware budget allocation, early head-type identification from consistent patterns, and stratified token eviction.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.18868","ref_index":32,"ref_count":1,"confidence":0.55,"is_internal_anchor":false,"paper_title":"DarkLLM: Learning Language-Driven Adversarial Attacks with Large Language Models","primary_cat":"cs.CR","submitted_at":"2026-05-15T12:28:16+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"DarkLLM trains an LLM to generate language-driven adversarial perturbations that unify targeted, untargeted, segmentation, and multi-model attacks on foundation models.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.15477","ref_index":46,"ref_count":2,"confidence":0.55,"is_internal_anchor":false,"paper_title":"EgoExo-WM: Unlocking Exo Video for Ego World Models","primary_cat":"cs.CV","submitted_at":"2026-05-14T23:35:54+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Method converts exocentric videos to egocentric format via body-pose extraction and kinematics to improve egocentric world-model prediction and planning.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.15398","ref_index":16,"ref_count":1,"confidence":0.55,"is_internal_anchor":false,"paper_title":"3DEditSafe: Defending 3D Editing Pipelines from Unsafe Generation","primary_cat":"cs.GR","submitted_at":"2026-05-14T20:30:50+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"3DEditSafe adds generation-stage guidance, 3D safety regularization, semantic projection, residue suppression, and mask-aware preservation to reduce unsafe semantic alignment in 3D editing while noting a safety-quality tradeoff.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.16423","ref_index":55,"ref_count":1,"confidence":0.55,"is_internal_anchor":false,"paper_title":"Nonlinear Bipolar Compensation: Handling Outliers in Post-Training Quantization","primary_cat":"cs.CV","submitted_at":"2026-05-14T14:55:46+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Nonlinear Bipolar Compensation with Bipolar Logarithmic Transformation reduces outlier effects in post-training quantization by performing compensation in a compressed transformed space.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.14773","ref_index":23,"ref_count":1,"confidence":0.55,"is_internal_anchor":false,"paper_title":"Beyond What to Select: A Plug-and-play Oscillatory Data-Volume Scheduling for Efficient Model Training","primary_cat":"cs.LG","submitted_at":"2026-05-14T12:37:11+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"PODS is a plug-and-play oscillatory data-volume scheduler that alternates low-ratio regularization phases with high-ratio recovery phases to improve data selection efficiency across training tasks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.13122","ref_index":18,"ref_count":1,"confidence":0.55,"is_internal_anchor":false,"paper_title":"Early Semantic Grounding in Image Editing Models for Zero-Shot Referring Image Segmentation","primary_cat":"cs.CV","submitted_at":"2026-05-13T07:48:05+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Pretrained instruction-based image editing models exhibit early foreground-background separability that enables a training-free framework for zero-shot referring image segmentation using a single denoising step.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.12122","ref_index":26,"ref_count":1,"confidence":0.55,"is_internal_anchor":false,"paper_title":"Disentangled Sparse Representations for Concept-Separated Diffusion Unlearning","primary_cat":"cs.LG","submitted_at":"2026-05-12T13:39:07+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"SAEParate disentangles sparse representations in diffusion models via contrastive clustering and nonlinear encoding to enable more precise concept unlearning with reduced side effects.","context_count":1,"top_context_role":"background","top_context_polarity":"background","context_text":"URLhttps://arxiv.org/abs/2502.04878. 11 [25] Senmao Li, Joost van de Weijer, taihang Hu, Fahad Khan, Qibin Hou, Yaxing Wang, and jian Yang. Get what you want, not what you don't: Image content suppression for text-to-image diffusion models. InThe Twelfth International Conference on Learning Representations, 2024. URLhttps://openreview.net/forum?id=zpVPhvVKXk. [26] Tsung-Yi Lin, Michael Maire, Serge Belongie, James Hays, Pietro Perona, Deva Ramanan, Piotr Dollár, and C Lawrence Zitnick. Microsoft coco: Common objects in context. InEuropean conference on computer vision, pages 740-755. Springer, 2014. [27] Mengyao Lyu, Yuhong Yang, Haiwen Hong, Hui Chen, Xuan Jin, Yuan He, Hui Xue, Jungong Han, and Guiguang Ding."},{"citing_arxiv_id":"2605.12088","ref_index":21,"ref_count":2,"confidence":0.55,"is_internal_anchor":false,"paper_title":"UniCustom: Unified Visual Conditioning for Multi-Reference Image Generation","primary_cat":"cs.CV","submitted_at":"2026-05-12T13:10:05+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"A unified visual conditioning approach fuses semantic and appearance features before VLM processing, with two-stage training and slot-wise regularization, to improve consistency in multi-reference image generation.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.12574","ref_index":29,"ref_count":1,"confidence":0.55,"is_internal_anchor":false,"paper_title":"DistractMIA: Black-Box Membership Inference on Vision-Language Models via Semantic Distraction","primary_cat":"cs.CV","submitted_at":"2026-05-12T12:04:11+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"DistractMIA performs output-only black-box membership inference on vision-language models by inserting semantic distractors and measuring shifts in generated text responses.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.11705","ref_index":31,"ref_count":1,"confidence":0.55,"is_internal_anchor":false,"paper_title":"CAST: Collapse-Aware multi-Scale Topology Fusion for Multimodal Coreset Selection","primary_cat":"cs.CV","submitted_at":"2026-05-12T07:59:08+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"CAST selects better multimodal coresets by fusing collapse-aware topologies across modalities and matching distributions at multiple scales in the diffusion wavelet domain.","context_count":1,"top_context_role":"dataset","top_context_polarity":"use_dataset","context_text":"Upon convergence, following FAST [8], we employ the Hungarian algorithm to map the optimized continuous proxy points back to authentic image-text pairs, thereby forming the final coresetC ⊂ D(details deferred to Appendix A.2). 4 Experiments 4.1 Experimental Setup Datasets and Metrics.Following [21] and [22], we evaluate our method on image captioning datasets: Flickr30K [30] and MS-COCO [31]. We adopt the Karpathy split [32], where Flickr30K contains 29k/1k/1k images and MS-COCO contains 113k/5k/5k images for training/validation/test. Each image is paired with five human-annotated captions. We evaluate bidirectional retrieval performance using Recall@K with K∈ {1,5,10} . Specifically, IR@K denotes text-to-image retrieval, and TR@K denotes image-to-text retrieval."},{"citing_arxiv_id":"2605.11563","ref_index":28,"ref_count":1,"confidence":0.55,"is_internal_anchor":false,"paper_title":"TCP-SSM: Efficient Vision State Space Models with Token-Conditioned Poles","primary_cat":"cs.CV","submitted_at":"2026-05-12T05:49:46+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"TCP-SSM conditions stable poles on visual tokens to explicitly control memory decay and oscillation in SSMs, cutting computation up to 44% while matching or exceeding accuracy on classification, segmentation, and detection.","context_count":1,"top_context_role":"dataset","top_context_polarity":"use_dataset","context_text":"is especially significant as dense prediction dictates much stricter demands on spatial granularity than standard image classification. The concurrent improvement in performance and reduction in computational overhead confirm that our token-conditioned pole mechanism does not compromise representational capacity for efficiency. 4.3 Object Detection and Instance Segmentation Settings:We evaluate TCP-SSM on COCO 2017 [ 28] for object detection and instance segmenta- tion within a ViTDet-style framework using Simple Feature Pyramid [25] and Cascade Mask R-CNN [5] heads. We replace only the Vim backbone with TCP-SSM and report box AP and mask AP. We provide detailed settings in Section??. Results:Table 3 presents object detection and instance segmentation results on COCO, demon-"},{"citing_arxiv_id":"2605.11497","ref_index":17,"ref_count":1,"confidence":0.55,"is_internal_anchor":false,"paper_title":"PoseBridge: Bridging the Skeletonization Gap for Zero-Shot Skeleton-Based Action Recognition","primary_cat":"cs.CV","submitted_at":"2026-05-12T04:15:52+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"PoseBridge recovers semantic information lost during skeletonization by extracting pose-anchored cues from human pose estimation and transferring them via skeleton-conditioned bridging and semantic prototype adaptation, yielding 13.3-17.4 point gains on the Kinetics PURLS benchmark.","context_count":1,"top_context_role":"dataset","top_context_polarity":"use_dataset","context_text":"Here, h, w index spatial locations in ˜FL, Wp projects the pooled feature into the semantic space, and ε is used for numerical stability. Body-aware pooling encourages p to emphasize visual evidence near the predicted body structure, including local appearance cues around action-relevant body parts. Action-semantic alignment. To align pose-anchored semantics with language, we use image-level captions from MS COCO [17] as weak image-language supervision. Each paired description d is encoded by a pretrained CLIP text encoder Ed and projected into the same embedding space, yielding t=W dEd(d). Given a batch of B pose-text pairs, we optimize a CLIP-style symmetric contrastive loss: LHPE sem =− 1 2B BX i=1 \" log exp(p⊤ i ti/τ)PB j=1 exp(p⊤ i tj/τ) + log exp(p⊤ i ti/τ)PB"},{"citing_arxiv_id":"2605.09591","ref_index":18,"ref_count":1,"confidence":0.55,"is_internal_anchor":false,"paper_title":"From Pixels to Concepts: Do Segmentation Models Understand What They Segment?","primary_cat":"cs.CV","submitted_at":"2026-05-10T15:07:17+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"CAFE benchmark reveals that promptable segmentation models often produce correct masks for misleading prompts, showing a gap between localization accuracy and true concept understanding.","context_count":1,"top_context_role":"dataset","top_context_polarity":"use_dataset","context_text":"coupling a grounding or detection model, such as Grounding DINO [20], with a mask generator [26]. Recently, SAM3 [1] introduced promptable concept segmentation (PCS), an end-to-end formulation that directly produces masks from concept prompts, without relying on an explicit grounding or detection stage to generate intermediate boxes. Standard benchmarks such as COCO [18], ADE20K [38], and LVIS [6] primarily evaluate segmentation accuracy over predefined visual categories. Recent counterfactual benchmarks, such as HalluSegBench [17] further tests object-level counterfactual hallucination by pairing factual images with counterfactual images in which the referred object is absent. However, counterfactual segmentation is not limited to object-level presence or absence."},{"citing_arxiv_id":"2605.07604","ref_index":25,"ref_count":1,"confidence":0.55,"is_internal_anchor":false,"paper_title":"SAM 3D Animal: Promptable Animal 3D Reconstruction from Images in the Wild","primary_cat":"cs.CV","submitted_at":"2026-05-08T11:26:46+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"SAM 3D Animal is the first promptable framework for multi-animal 3D reconstruction from single images, built on SMAL+ and trained on the new Herd3D dataset, achieving SOTA results on Animal3D, APTv2, and Animal Kingdom benchmarks.","context_count":1,"top_context_role":"method","top_context_polarity":"use_method","context_text":"GenZoo [30], which builds upon the SMAL+ variant. Additionally, we include 3D Fauna [24] as a representative SOTA model-free reconstruction approach. Evaluation Metrics.We evaluate 3D accuracy using the Procrustes-Aligned Mean Per Joint Position Error (PA-MPJPE). For 2D accuracy, we report the Percentage of Correct Keypoints (PCK), AP (Average Precision) and mAP (mean Average Precision) [25]. Implementation Details.Our network is optimized using AdamW [ 26] with an initial learning rate of 2×10 −5, incorporating a linear warmup over the first 15 epochs. Similar to AniMer [ 28], we employ a two-stage training strategy consisting of 250 epochs for the first stage and 250 epochs for the second. We apply prompt dropout for robustness: the mask prompt is dropped with 50%"},{"citing_arxiv_id":"2605.07402","ref_index":40,"ref_count":1,"confidence":0.55,"is_internal_anchor":false,"paper_title":"InsHuman: Towards Natural and Identity-Preserving Human Insertion","primary_cat":"cs.CV","submitted_at":"2026-05-08T07:58:03+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"InsHuman proposes Human-Background Adaptive Fusion, Face-to-Face ID-Preserving, and Bidirectional Data Pairing to enable natural human insertion in images without altering identity.","context_count":1,"top_context_role":"background","top_context_polarity":"background","context_text":"person count control, and facial identity preservation within a unified training strategy. Human-Centric Datasets.Mainstream portrait datasets [ 29, 28, 37] focus on head or half-body close-ups, lacking full-body and scene context. DeepFashion [ 30], StyleGAN-Human [ 38], and Text2Human [39] provide high-resolution full-body images but mostly feature plain studio back- grounds. Natural scene datasets such as MS COCO [40] and CrowdHuman [31] offer environmental diversity but suffer from severe occlusions or low resolution. Meanwhile, recent compositional editing benchmarks [41, 42] primarily evaluate general object relations, lacking a specific focus on high- fidelity person-scene physical interactions. Recent generation-focused datasets like HumanSD [23] improve structural diversity but still lack paired input-output images required for scene insertion."},{"citing_arxiv_id":"2605.06376","ref_index":22,"ref_count":1,"confidence":0.55,"is_internal_anchor":false,"paper_title":"Continuous-Time Distribution Matching for Few-Step Diffusion Distillation","primary_cat":"cs.CV","submitted_at":"2026-05-07T14:56:39+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":8.0,"formal_verification":"none","one_line_summary":"CDM migrates distribution matching distillation to continuous time via dynamic random-length schedules and active off-trajectory latent alignment, yielding competitive few-step image fidelity on SD3 and Longcat-Image.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.03456","ref_index":27,"ref_count":3,"confidence":0.55,"is_internal_anchor":false,"paper_title":"VL-SAM-v3: Memory-Guided Visual Priors for Open-World Object Detection","primary_cat":"cs.CV","submitted_at":"2026-05-05T07:44:03+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"VL-SAM-v3 retrieves visual prototypes from memory to generate sparse spatial and dense contextual priors that refine detection prompts, yielding gains on rare categories in LVIS for both open-vocabulary and open-ended settings.","context_count":1,"top_context_role":"dataset","top_context_polarity":"use_dataset","context_text":"open-ended detection, Table 2 explicitly reports the base detector and candidate generator for each method. Rows with the same base detector and candidate generator provide controlled comparisons that isolate detector-side improvements, while rows with stronger candidate generators examine the complementarity between category generation and retrieval-grounded refinement. Following LLMDet [11], we also report COCO [ 27] results for reference, although these are not zero-shot because GroundingCap-1M contains COCO images. Implementation details.We instantiate VL-SAM-v3 with two base open-vocabulary detectors. Our main implementation builds on LLMDet [11], and initializes the Swin-T and Swin-L variants from the corresponding pretrained checkpoints. We follow LLMDet for the training strategy and loss"},{"citing_arxiv_id":"2605.01766","ref_index":21,"ref_count":1,"confidence":0.55,"is_internal_anchor":false,"paper_title":"Mitigating Multimodal LLMs Hallucinations via Relevance Propagation at Inference Time","primary_cat":"cs.LG","submitted_at":"2026-05-03T07:58:02+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"LIME reduces hallucinations in multimodal LLMs by using LRP to boost perceptual modality contributions through inference-time KV updates.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.01711","ref_index":22,"ref_count":1,"confidence":0.55,"is_internal_anchor":false,"paper_title":"Linear-Time Global Visual Modeling without Explicit Attention","primary_cat":"cs.CV","submitted_at":"2026-05-03T04:51:30+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Dynamic parameterization of standard layers can replace explicit attention for linear-time global visual modeling.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.01330","ref_index":26,"ref_count":1,"confidence":0.55,"is_internal_anchor":false,"paper_title":"Colinearity Decay: Training Quantization-Friendly ViTs with Outlier Decay","primary_cat":"cs.CV","submitted_at":"2026-05-02T08:49:51+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Colinearity-Decay regularizer trains ViTs that maintain or improve full-precision accuracy while delivering higher accuracy after low-bit quantization on ImageNet and COCO tasks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.13710","ref_index":23,"ref_count":2,"confidence":0.55,"is_internal_anchor":false,"paper_title":"SLQ: Bridging Modalities via Shared Latent Queries for Retrieval with Frozen MLLMs","primary_cat":"cs.CV","submitted_at":"2026-04-15T10:39:42+00:00","verdict":"CONDITIONAL","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"SLQ adapts frozen MLLMs for multimodal retrieval by appending shared latent queries to text and image tokens and introduces KARR-Bench to test knowledge-aware reasoning retrieval.","context_count":1,"top_context_role":"dataset","top_context_polarity":"use_dataset","context_text":"SLQ-8B* 41.1 10.5 47.6 59.7 37.8 35.5 36.8 SLQ-1B† 52.5 49.1 60.2 78.3 59.8 54.5 57.3 SLQ-2B† 54.2 53.3 62.1 79.5 62.3 56.0 59.6 SLQ-4B† 55.7 58.7 65.4 84.2 66.5 58.6 64.5 SLQ-8B† 60.9 61.270.5 86.669.4 61.7 67.5 5 Experiments Evaluation Benchmarks.We evaluate our method on a diverse set of benchmarks. For standard retrieval, we use Flickr30K [31] and COCO [23]. We further evaluate on MMEB [15], which consists of 36 tasks, to assess general multimodal embedding capability across tasks, and on our proposed KARR-Bench to measure knowledge-aware reasoning ability. Implementation Details.We use InternVL3 (1B, 8B) [ 50] and Qwen3-VL (2B, 4B) [4] as back- bones. To preserve pre-trained capabilities, the entire backbone is frozen during training, and only"},{"citing_arxiv_id":"2512.21788","ref_index":19,"ref_count":1,"confidence":0.55,"is_internal_anchor":false,"paper_title":"InstructMoLE: Instruction-Guided Mixture of Low-rank Experts for Multi-Conditional Image Generation","primary_cat":"cs.CV","submitted_at":"2025-12-25T21:37:12+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"InstructMoLE replaces per-token routing with instruction-guided global routing for mixture-of-low-rank-experts in diffusion transformers and adds an output-space orthogonality loss to improve multi-conditional image generation.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2510.13768","ref_index":46,"ref_count":1,"confidence":0.55,"is_internal_anchor":false,"paper_title":"Scaling Vision Transformers for Functional MRI with Flat Maps","primary_cat":"cs.CV","submitted_at":"2025-10-15T17:15:00+00:00","verdict":"CONDITIONAL","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"CortexMAE adapts Vision Transformers to fMRI via cortical flat maps, shows power-law scaling on 2.1K hours of data, and outperforms priors on cognitive state decoding while failing to beat a simple functional connectivity baseline on subject-level trait prediction.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2210.08402","ref_index":46,"ref_count":1,"confidence":0.55,"is_internal_anchor":false,"paper_title":"LAION-5B: An open large-scale dataset for training next generation image-text models","primary_cat":"cs.CV","submitted_at":"2022-10-16T00:08:18+00:00","verdict":"ACCEPT","verdict_confidence":"MODERATE","novelty_score":7.0,"formal_verification":"none","one_line_summary":"LAION-5B is an openly released dataset of 5.85 billion CLIP-filtered image-text pairs that enables replication of foundational vision-language models.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null}],"limit":50,"offset":0}