{"id":"127576e8-3ad0-473f-b767-c37341d4e045","arxiv_id":"2505.13972","paper_version":3,"verdict":"CONDITIONAL","confidence":"MODERATE","novelty_score":6.0,"correctness_risk":"medium","formal_verification":"none","parameter_count":0,"one_line_summary":"Independent, non-fine-tuned judge models align most closely with human label-flip judgments, yet all automated judges fall well short of human evaluation.","lead":"This paper tests which AI judge model should evaluate whether counterfactual examples have flipped their ground-truth label in counterfactual data augmentation. It finds that judges independent of the generator and not fine-tuned on the dataset agree best with human annotators, but even the best automated judge remains far from human judgment.","discovery_kind":"new_application","skeptic_critique":null,"referee_report":null,"author_rebuttal":null,"desk_editor":null,"rs_alignment":null,"lean_confirmation":null,"pith_extraction":null,"created_at":"2026-08-07T15:43:52.916300+00:00","model_set":{"reader":"deepseek-v4-flash"},"falsifier":null,"supporting_citations":[],"review_version":1}