{"id":"3f57996f-7f0a-448f-a053-69d01424229e","arxiv_id":"2509.24901","paper_version":4,"verdict":"CONDITIONAL","confidence":"MODERATE","novelty_score":6.0,"correctness_risk":"medium","formal_verification":"none","parameter_count":2,"one_line_summary":"A binarized prototypical probing method that pools per-class evidence from patch tokens substantially outperforms [cls]-token and attentive probes on multi-label audio classification benchmarks.","lead":"This paper argues that frozen audio AI models are better evaluated with per-class prototype probes over the full token map than with the standard [cls]-token or attention pooling. The proposed binarized prototypical probe beats existing lightweight probes across 13 datasets and could make fine-tuning less necessary for model comparison.","discovery_kind":"new_method","skeptic_critique":null,"referee_report":null,"author_rebuttal":null,"desk_editor":null,"rs_alignment":null,"lean_confirmation":null,"pith_extraction":null,"created_at":"2026-08-04T13:51:34.083998+00:00","model_set":{"reader":"deepseek-v4-flash"},"falsifier":null,"supporting_citations":[],"review_version":1}