{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:VXRAOCJ6H3Q7TUJTCT2H3WEJXH","short_pith_number":"pith:VXRAOCJ6","schema_version":"1.0","canonical_sha256":"ade207093e3ee1f9d13314f47dd889b9e5721aac0b0a288184dd5d137e6cfc30","source":{"kind":"arxiv","id":"2504.17087","version":1},"attestation_state":"computed","paper":{"title":"Leveraging LLMs as Meta-Judges: A Multi-Agent Framework for Evaluating LLM Judgments","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Benoit Boulet, Chongren Sun, Di Wu, Jama Hussein Mohamud, Yuran Li","submitted_at":"2025-04-23T20:32:12Z","abstract_excerpt":"Large language models (LLMs) are being widely applied across various fields, but as tasks become more complex, evaluating their responses is increasingly challenging. Compared to human evaluators, the use of LLMs to support performance evaluation offers a more efficient alternative. However, most studies focus mainly on aligning LLMs' judgments with human preferences, overlooking the existence of biases and mistakes in human judgment. Furthermore, how to select suitable LLM judgments given multiple potential LLM responses remains underexplored. To address these two aforementioned issues, we pr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.17087","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2025-04-23T20:32:12Z","cross_cats_sorted":[],"title_canon_sha256":"8830b7f5f5a64cf73696bffd04ab0d6d66000cfafdb344b7e1c3e07c4ace38ed","abstract_canon_sha256":"54b9676a0756595d71995e43fdb8bb7f9658178cec6563e7f67afd34a108faff"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:53:22.057205Z","signature_b64":"ND9wiNsibiUYQVWnr8TOyjHHjT+7rZJR5mAd5J3TWq8nDrwRaQDKEOTrw7Ac21ph7R+2Bklf0uwrugE0s1lFAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ade207093e3ee1f9d13314f47dd889b9e5721aac0b0a288184dd5d137e6cfc30","last_reissued_at":"2026-07-05T10:53:22.056697Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:53:22.056697Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Leveraging LLMs as Meta-Judges: A Multi-Agent Framework for Evaluating LLM Judgments","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Benoit Boulet, Chongren Sun, Di Wu, Jama Hussein Mohamud, Yuran Li","submitted_at":"2025-04-23T20:32:12Z","abstract_excerpt":"Large language models (LLMs) are being widely applied across various fields, but as tasks become more complex, evaluating their responses is increasingly challenging. Compared to human evaluators, the use of LLMs to support performance evaluation offers a more efficient alternative. However, most studies focus mainly on aligning LLMs' judgments with human preferences, overlooking the existence of biases and mistakes in human judgment. Furthermore, how to select suitable LLM judgments given multiple potential LLM responses remains underexplored. To address these two aforementioned issues, we pr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.17087","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.17087/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.17087","created_at":"2026-07-05T10:53:22.056762+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.17087v1","created_at":"2026-07-05T10:53:22.056762+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.17087","created_at":"2026-07-05T10:53:22.056762+00:00"},{"alias_kind":"pith_short_12","alias_value":"VXRAOCJ6H3Q7","created_at":"2026-07-05T10:53:22.056762+00:00"},{"alias_kind":"pith_short_16","alias_value":"VXRAOCJ6H3Q7TUJT","created_at":"2026-07-05T10:53:22.056762+00:00"},{"alias_kind":"pith_short_8","alias_value":"VXRAOCJ6","created_at":"2026-07-05T10:53:22.056762+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21627","citing_title":"Counsel: A Meta-Evaluation Dataset for Agentic Tasks","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03650","citing_title":"CoEval: Ranking Language Models for Custom Tasks Without Labeled Data or Trustworthy Benchmarks","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02359","citing_title":"Using LLM-as-a-Judge/Jury to Advance Scalable, Clinically-Validated Safety Evaluations of Model Responses to Users Demonstrating Psychosis","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09702","citing_title":"Calibrate, Don't Curate: Label-Efficient Estimation from Noisy LLM Judges","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VXRAOCJ6H3Q7TUJTCT2H3WEJXH","json":"https://pith.science/pith/VXRAOCJ6H3Q7TUJTCT2H3WEJXH.json","graph_json":"https://pith.science/api/pith-number/VXRAOCJ6H3Q7TUJTCT2H3WEJXH/graph.json","events_json":"https://pith.science/api/pith-number/VXRAOCJ6H3Q7TUJTCT2H3WEJXH/events.json","paper":"https://pith.science/paper/VXRAOCJ6"},"agent_actions":{"view_html":"https://pith.science/pith/VXRAOCJ6H3Q7TUJTCT2H3WEJXH","download_json":"https://pith.science/pith/VXRAOCJ6H3Q7TUJTCT2H3WEJXH.json","view_paper":"https://pith.science/paper/VXRAOCJ6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.17087&json=true","fetch_graph":"https://pith.science/api/pith-number/VXRAOCJ6H3Q7TUJTCT2H3WEJXH/graph.json","fetch_events":"https://pith.science/api/pith-number/VXRAOCJ6H3Q7TUJTCT2H3WEJXH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VXRAOCJ6H3Q7TUJTCT2H3WEJXH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VXRAOCJ6H3Q7TUJTCT2H3WEJXH/action/storage_attestation","attest_author":"https://pith.science/pith/VXRAOCJ6H3Q7TUJTCT2H3WEJXH/action/author_attestation","sign_citation":"https://pith.science/pith/VXRAOCJ6H3Q7TUJTCT2H3WEJXH/action/citation_signature","submit_replication":"https://pith.science/pith/VXRAOCJ6H3Q7TUJTCT2H3WEJXH/action/replication_record"}},"created_at":"2026-07-05T10:53:22.056762+00:00","updated_at":"2026-07-05T10:53:22.056762+00:00"}