{"total":14,"items":[{"citing_arxiv_id":"2607.08674","ref_index":28,"ref_count":1,"confidence":0.35,"is_internal_anchor":false,"paper_title":"Do Transformations Reveal the Truth? Generative Residual Learning for Generalized AI-Generated Image Detection","primary_cat":"cs.CV","submitted_at":"2026-07-09T16:36:54+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Modeling the differential response of real versus AI-generated images under secondary generative transformations yields state-of-the-art cross-generator detection on 19 unseen models.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.00517","ref_index":27,"ref_count":1,"confidence":0.35,"is_internal_anchor":false,"paper_title":"PhysiGen: Integrating Collision-Aware Physical Constraints for High-Fidelity Human-Human Interaction Generation","primary_cat":"cs.CV","submitted_at":"2026-05-01T08:43:16+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"PhysiGen reduces interpenetration in text-driven 3D human interaction generation by simplifying meshes to geometric primitives for fast collision detection and guiding optimization with collision regions.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.22990","ref_index":18,"ref_count":1,"confidence":0.35,"is_internal_anchor":false,"paper_title":"Hard to See, Hard to Label: Generative and Symbolic Acquisition for Subtle Visual Phenomena","primary_cat":"cs.CV","submitted_at":"2026-04-24T20:05:41+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"GSAL combines diffusion-based visual difficulty scoring with hierarchical semantic coverage to improve active learning retrieval of subtle and rare visual anomalies over standard uncertainty and diversity methods.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.21453","ref_index":31,"ref_count":1,"confidence":0.35,"is_internal_anchor":false,"paper_title":"Instance-level Visual Active Tracking with Occlusion-Aware Planning","primary_cat":"cs.CV","submitted_at":"2026-04-23T09:11:50+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"OA-VAT improves visual active tracking by combining instance-level prototype discrimination with occlusion-aware diffusion planning, reporting gains over prior SOTA on simulated and real drone benchmarks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.19406","ref_index":33,"ref_count":1,"confidence":0.35,"is_internal_anchor":false,"paper_title":"HP-Edit: A Human-Preference Post-Training Framework for Image Editing","primary_cat":"cs.CV","submitted_at":"2026-04-21T12:29:50+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"HP-Edit introduces a post-training framework and RealPref-50K dataset that uses a VLM-based HP-Scorer to align diffusion image editing models with human preferences, improving outputs on Qwen-Image-Edit-2509.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.19141","ref_index":43,"ref_count":1,"confidence":0.35,"is_internal_anchor":false,"paper_title":"Denoising, Fast and Slow: Difficulty-Aware Adaptive Sampling for Image Generation","primary_cat":"cs.CV","submitted_at":"2026-04-21T06:44:44+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Patch Forcing enables diffusion models to denoise image patches at varying rates based on predicted difficulty, advancing easier regions first to improve context and achieve better generation quality on ImageNet while scaling to text-to-image tasks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.13581","ref_index":38,"ref_count":1,"confidence":0.35,"is_internal_anchor":false,"paper_title":"SocialMirror: Reconstructing 3D Human Interaction Behaviors from Monocular Videos with Semantic and Geometric Guidance","primary_cat":"cs.CV","submitted_at":"2026-04-15T07:41:52+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"SocialMirror reconstructs 3D meshes of closely interacting humans from monocular videos using semantic guidance from vision-language models and geometric constraints in a diffusion model to handle occlusions and maintain temporal and spatial consistency.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.13305","ref_index":46,"ref_count":1,"confidence":0.35,"is_internal_anchor":false,"paper_title":"Bias at the End of the Score","primary_cat":"cs.CV","submitted_at":"2026-04-14T21:20:47+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Reward models used as quality scorers in text-to-image generation encode demographic biases that cause reward-guided training to sexualize female subjects, reinforce stereotypes, and reduce diversity.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.11470","ref_index":24,"ref_count":1,"confidence":0.35,"is_internal_anchor":false,"paper_title":"Degradation-Aware and Structure-Preserving Diffusion for Real-World Image Super-Resolution","primary_cat":"cs.CV","submitted_at":"2026-04-13T13:43:22+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":4.0,"formal_verification":"none","one_line_summary":"Two new lightweight modules for diffusion-based real-world image super-resolution deliver competitive perceptual quality and better structure preservation on DIV2K and RealSR datasets.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.09405","ref_index":35,"ref_count":1,"confidence":0.35,"is_internal_anchor":false,"paper_title":"EGLOCE: Training-Free Energy-Guided Latent Optimization for Concept Erasure","primary_cat":"cs.CV","submitted_at":"2026-04-10T15:19:02+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"EGLOCE erases target concepts in diffusion models at inference time by optimizing latents with dual energy guidance that repels unwanted concepts while retaining prompt alignment.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2603.26588","ref_index":15,"ref_count":1,"confidence":0.35,"is_internal_anchor":false,"paper_title":"From Synthetic Data to Real Restorations: Diffusion Model for Patient-specific Dental Crown Completion","primary_cat":"cs.CV","submitted_at":"2026-03-27T16:51:40+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"A diffusion model trained on synthetically damaged teeth from public datasets completes crowns with 81.8% IoU and 0.00034 Chamfer distance, and produces real-world restorations with minimal opposing-tooth interference.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2602.21581","ref_index":20,"ref_count":1,"confidence":0.35,"is_internal_anchor":false,"paper_title":"MultiAnimate: Pose-Guided Image Animation Made Extensible","primary_cat":"cs.CV","submitted_at":"2026-02-25T05:06:58+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"MultiAnimate adds Identifier Assigner and Identifier Adapter modules to diffusion video models so they can handle multiple characters without identity mix-ups, generalizing from two-character training data to more characters.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2512.04678","ref_index":52,"ref_count":1,"confidence":0.35,"is_internal_anchor":false,"paper_title":"Reward Forcing: Efficient Streaming Video Generation with Rewarded Distribution Matching Distillation","primary_cat":"cs.CV","submitted_at":"2025-12-04T11:12:13+00:00","verdict":"CONDITIONAL","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Reward Forcing combines EMA-Sink tokens and Rewarded Distribution Matching Distillation to deliver state-of-the-art streaming video generation at 23.1 FPS without copying initial frames.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2505.18991","ref_index":36,"ref_count":1,"confidence":0.35,"is_internal_anchor":false,"paper_title":"Fast Kernel-Space Diffusion for Remote Sensing Pansharpening","primary_cat":"cs.CV","submitted_at":"2025-05-25T06:25:31+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"KSDiff generates convolutional kernels in kernel space using low-rank core tensor and factor generators with multi-head attention for fast, high-quality pansharpening.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null}],"limit":50,"offset":0}