{"work":{"id":"2e127924-0ccf-418b-9de5-0daa4fb9ad9b","openalex_id":null,"doi":null,"arxiv_id":"2410.10934","raw_key":null,"title":"Agent-as-a-Judge: Evaluate Agents with Agents","authors":null,"authors_text":"Agent-as-a-judge: Evaluate agents with agents , author=","year":2024,"venue":"cs.AI","abstract":"Contemporary evaluation techniques are inadequate for agentic systems. These approaches either focus exclusively on final outcomes -- ignoring the step-by-step nature of agentic systems, or require excessive manual labour. To address this, we introduce the Agent-as-a-Judge framework, wherein agentic systems are used to evaluate agentic systems. This is an organic extension of the LLM-as-a-Judge framework, incorporating agentic features that enable intermediate feedback for the entire task-solving process. We apply the Agent-as-a-Judge to the task of code generation. To overcome issues with existing benchmarks and provide a proof-of-concept testbed for Agent-as-a-Judge, we present DevAI, a new benchmark of 55 realistic automated AI development tasks. It includes rich manual annotations, like a total of 365 hierarchical user requirements. We benchmark three of the popular agentic systems using Agent-as-a-Judge and find it dramatically outperforms LLM-as-a-Judge and is as reliable as our human evaluation baseline. Altogether, we believe that Agent-as-a-Judge marks a concrete step forward for modern agentic systems -- by providing rich and reliable reward signals necessary for dynamic and scalable self-improvement.","external_url":"https://arxiv.org/abs/2410.10934","cited_by_count":null,"metadata_source":"pith","metadata_fetched_at":"2026-07-10T15:07:19.644728+00:00","pith_arxiv_id":"2410.10934","created_at":"2026-05-10T22:15:50.729603+00:00","updated_at":"2026-07-10T15:07:19.644728+00:00","title_quality_ok":true,"display_title":"arXiv preprint 2410.10934","render_title":"arXiv preprint 2410.10934"},"hub":{"state":{"work_id":"2e127924-0ccf-418b-9de5-0daa4fb9ad9b","tier":"hub","tier_reason":"10+ Pith inbound or 1,000+ external citations","pith_inbound_count":28,"external_cited_by_count":null,"distinct_field_count":6,"first_pith_cited_at":"2024-11-23T16:03:35+00:00","last_pith_cited_at":"2026-07-08T21:45:34+00:00","author_build_status":"not_needed","summary_status":"needed","contexts_status":"needed","graph_status":"needed","ask_index_status":"not_needed","reader_status":"not_needed","recognition_status":"not_needed","updated_at":"2026-08-22T07:09:34.654395+00:00","tier_text":"hub"},"tier":"hub","role_counts":[{"context_role":"background","n":4},{"context_role":"dataset","n":1}],"polarity_counts":[{"context_polarity":"background","n":4},{"context_polarity":"use_dataset","n":1}],"runs":{},"summary":{},"graph":{},"authors":[]}}