{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:5UHH7F2SS4G5OC3SOP3WXZ4N2G","short_pith_number":"pith:5UHH7F2S","schema_version":"1.0","canonical_sha256":"ed0e7f9752970dd70b7273f76be78dd1a0302bd1640fb692da86e0ebdb103a39","source":{"kind":"arxiv","id":"2406.18403","version":3},"attestation_state":"computed","paper":{"title":"LLMs instead of Human Judges? A Large Scale Empirical Study across 20 NLP Evaluation Tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Aditya K Surikuchi, Albert Gatt, Alberto Testoni, Alessandro Suglia, Alexander Koller, Andr\\'e F. T. Martins, Anna Bavaresco, Barbara Plank, David Schlangen, Desmond Elliott, Ece Takmaz, Esam Ghaleb, Leonardo Bertolazzi, Mario Giulianelli, Michael Hanna, Philipp Mondorf, Raffaella Bernardi, Raquel Fern\\'andez, Sandro Pezzelle, Vera Neplenbroek","submitted_at":"2024-06-26T14:56:13Z","abstract_excerpt":"There is an increasing trend towards evaluating NLP models with LLMs instead of human judgments, raising questions about the validity of these evaluations, as well as their reproducibility in the case of proprietary models. We provide JUDGE-BENCH, an extensible collection of 20 NLP datasets with human annotations covering a broad range of evaluated properties and types of data, and comprehensively evaluate 11 current LLMs, covering both open-weight and proprietary models, for their ability to replicate the annotations. Our evaluations show substantial variance across models and datasets. Model"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.18403","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-26T14:56:13Z","cross_cats_sorted":[],"title_canon_sha256":"7fbf834f830fea0fcd14b0265258a361185cbcb357440656136358bb638f6e27","abstract_canon_sha256":"b6fdb8abe8782f8d0e350301a1ed31fdc97aae6e93b3d1d766b3be2ae8c37e4a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:13:56.588637Z","signature_b64":"62rU8KKouNYY4woIWau6feU1PasIo8USjhvLBK5Moj/CoqPDvMO/VNcz3L2eli0YNUxv4hc6oqe5IFBc4JfaDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ed0e7f9752970dd70b7273f76be78dd1a0302bd1640fb692da86e0ebdb103a39","last_reissued_at":"2026-07-05T11:13:56.588127Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:13:56.588127Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LLMs instead of Human Judges? A Large Scale Empirical Study across 20 NLP Evaluation Tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Aditya K Surikuchi, Albert Gatt, Alberto Testoni, Alessandro Suglia, Alexander Koller, Andr\\'e F. T. Martins, Anna Bavaresco, Barbara Plank, David Schlangen, Desmond Elliott, Ece Takmaz, Esam Ghaleb, Leonardo Bertolazzi, Mario Giulianelli, Michael Hanna, Philipp Mondorf, Raffaella Bernardi, Raquel Fern\\'andez, Sandro Pezzelle, Vera Neplenbroek","submitted_at":"2024-06-26T14:56:13Z","abstract_excerpt":"There is an increasing trend towards evaluating NLP models with LLMs instead of human judgments, raising questions about the validity of these evaluations, as well as their reproducibility in the case of proprietary models. We provide JUDGE-BENCH, an extensible collection of 20 NLP datasets with human annotations covering a broad range of evaluated properties and types of data, and comprehensively evaluate 11 current LLMs, covering both open-weight and proprietary models, for their ability to replicate the annotations. Our evaluations show substantial variance across models and datasets. Model"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.18403","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.18403/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.18403","created_at":"2026-07-05T11:13:56.588192+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.18403v3","created_at":"2026-07-05T11:13:56.588192+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.18403","created_at":"2026-07-05T11:13:56.588192+00:00"},{"alias_kind":"pith_short_12","alias_value":"5UHH7F2SS4G5","created_at":"2026-07-05T11:13:56.588192+00:00"},{"alias_kind":"pith_short_16","alias_value":"5UHH7F2SS4G5OC3S","created_at":"2026-07-05T11:13:56.588192+00:00"},{"alias_kind":"pith_short_8","alias_value":"5UHH7F2S","created_at":"2026-07-05T11:13:56.588192+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12433","citing_title":"Marginal Alignment Does Not Guarantee Joint-Distribution Fidelity: An Official-Reference Audit of Nemotron-Personas-Korea with Cross-Locale Replication","ref_index":73,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22247","citing_title":"Natural Language-Focused Software Engineering via Code-Documentation Equivalence","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2508.15503","citing_title":"Guidelines for Empirical Studies in Software Engineering involving Large Language Models","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2508.15503","citing_title":"Guidelines for Empirical Studies in Software Engineering involving Large Language Models","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03246","citing_title":"Personalized AI Practice Replicates Learning Rate Regularity at Scale","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11206","citing_title":"Instructions Shape Production of Language, not Processing","ref_index":181,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11206","citing_title":"Instructions Shape Production of Language, not Processing","ref_index":181,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06856","citing_title":"Benchmarked Yet Not Measured -- Generative AI Should be Evaluated Against Real-World Utility","ref_index":221,"is_internal_anchor":false},{"citing_arxiv_id":"2412.05579","citing_title":"LLMs-as-Judges: A Comprehensive Survey on LLM-based Evaluation Methods","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06856","citing_title":"Benchmarked Yet Not Measured -- Generative AI Should be Evaluated Against Real-World Utility","ref_index":221,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16706","citing_title":"Evaluating Tool-Using Language Agents: Judge Reliability, Propagation Cascades, and Runtime Mitigation in AgentProp-Bench","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5UHH7F2SS4G5OC3SOP3WXZ4N2G","json":"https://pith.science/pith/5UHH7F2SS4G5OC3SOP3WXZ4N2G.json","graph_json":"https://pith.science/api/pith-number/5UHH7F2SS4G5OC3SOP3WXZ4N2G/graph.json","events_json":"https://pith.science/api/pith-number/5UHH7F2SS4G5OC3SOP3WXZ4N2G/events.json","paper":"https://pith.science/paper/5UHH7F2S"},"agent_actions":{"view_html":"https://pith.science/pith/5UHH7F2SS4G5OC3SOP3WXZ4N2G","download_json":"https://pith.science/pith/5UHH7F2SS4G5OC3SOP3WXZ4N2G.json","view_paper":"https://pith.science/paper/5UHH7F2S","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.18403&json=true","fetch_graph":"https://pith.science/api/pith-number/5UHH7F2SS4G5OC3SOP3WXZ4N2G/graph.json","fetch_events":"https://pith.science/api/pith-number/5UHH7F2SS4G5OC3SOP3WXZ4N2G/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5UHH7F2SS4G5OC3SOP3WXZ4N2G/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5UHH7F2SS4G5OC3SOP3WXZ4N2G/action/storage_attestation","attest_author":"https://pith.science/pith/5UHH7F2SS4G5OC3SOP3WXZ4N2G/action/author_attestation","sign_citation":"https://pith.science/pith/5UHH7F2SS4G5OC3SOP3WXZ4N2G/action/citation_signature","submit_replication":"https://pith.science/pith/5UHH7F2SS4G5OC3SOP3WXZ4N2G/action/replication_record"}},"created_at":"2026-07-05T11:13:56.588192+00:00","updated_at":"2026-07-05T11:13:56.588192+00:00"}