{"id":"ebc92a66-db94-4781-aad8-e5ee816b5e94","arxiv_id":"2504.12018","paper_version":1,"verdict":"CONDITIONAL","confidence":"MODERATE","novelty_score":5.0,"correctness_risk":"medium","formal_verification":"none","parameter_count":4,"one_line_summary":"A fine-tuned multimodal score model with soft Q-Align scoring, element-conditioned prompts, and self-training on validation pseudo-labels takes first place in NTIRE 2025 Track 1 image-text alignment.","lead":"iMatch is a method for automatically scoring how well an AI-generated image matches its text prompt, by fine-tuning multimodal language models with extra element labels, image augmentations, and pseudo-labeled retraining. A generalist might read this because reliable automatic scoring of text-to-image output is a practical need for evaluating and improving generative models.","discovery_kind":"extension","skeptic_critique":null,"referee_report":null,"author_rebuttal":null,"desk_editor":null,"rs_alignment":null,"lean_confirmation":null,"pith_extraction":null,"created_at":"2026-08-16T12:40:21.499652+00:00","model_set":{"reader":"deepseek-v4-flash"},"falsifier":null,"supporting_citations":[],"review_version":1}