{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:V26VZM4JY2DZ4P6GZPWNJQHIMR","short_pith_number":"pith:V26VZM4J","schema_version":"1.0","canonical_sha256":"aebd5cb389c6879e3fc6cbecd4c0e8644e80febe0afc09986c61c1f31286a97f","source":{"kind":"arxiv","id":"2410.20266","version":1},"attestation_state":"computed","paper":{"title":"Limitations of the LLM-as-a-Judge Approach for Evaluating LLM Outputs in Expert Knowledge Tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.HC","authors_text":"Annalisa Szymanski, Heather A. Eicher-Miller, Meng Jiang, Noah Ziems, Ronald A. Metoyer, Toby Jia-Jun Li","submitted_at":"2024-10-26T20:35:14Z","abstract_excerpt":"The potential of using Large Language Models (LLMs) themselves to evaluate LLM outputs offers a promising method for assessing model performance across various contexts. Previous research indicates that LLM-as-a-judge exhibits a strong correlation with human judges in the context of general instruction following. However, for instructions that require specialized knowledge, the validity of using LLMs as judges remains uncertain. In our study, we applied a mixed-methods approach, conducting pairwise comparisons in which both subject matter experts (SMEs) and LLMs evaluated outputs from domain-s"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.20266","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.HC","submitted_at":"2024-10-26T20:35:14Z","cross_cats_sorted":[],"title_canon_sha256":"3b74af65934b2ea1d05a46b3f55a093df872392dc1bb4f4aefab536816fd0b91","abstract_canon_sha256":"21dfc0fa27f7e92fa90f4ebc5fc206c49949742502cf2b0bf07da6bdfbb3bace"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:26:41.714286Z","signature_b64":"PhOG5MK7H49Z4afNWgcpcqQZccf7J33qiJ2ILeyXBBuXy15KLoIdXMHW7wY5w8GdXo+i5SfQPiXsvPUt7kjzAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"aebd5cb389c6879e3fc6cbecd4c0e8644e80febe0afc09986c61c1f31286a97f","last_reissued_at":"2026-07-05T09:26:41.713787Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:26:41.713787Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Limitations of the LLM-as-a-Judge Approach for Evaluating LLM Outputs in Expert Knowledge Tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.HC","authors_text":"Annalisa Szymanski, Heather A. Eicher-Miller, Meng Jiang, Noah Ziems, Ronald A. Metoyer, Toby Jia-Jun Li","submitted_at":"2024-10-26T20:35:14Z","abstract_excerpt":"The potential of using Large Language Models (LLMs) themselves to evaluate LLM outputs offers a promising method for assessing model performance across various contexts. Previous research indicates that LLM-as-a-judge exhibits a strong correlation with human judges in the context of general instruction following. However, for instructions that require specialized knowledge, the validity of using LLMs as judges remains uncertain. In our study, we applied a mixed-methods approach, conducting pairwise comparisons in which both subject matter experts (SMEs) and LLMs evaluated outputs from domain-s"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.20266","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.20266/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.20266","created_at":"2026-07-05T09:26:41.713853+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.20266v1","created_at":"2026-07-05T09:26:41.713853+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.20266","created_at":"2026-07-05T09:26:41.713853+00:00"},{"alias_kind":"pith_short_12","alias_value":"V26VZM4JY2DZ","created_at":"2026-07-05T09:26:41.713853+00:00"},{"alias_kind":"pith_short_16","alias_value":"V26VZM4JY2DZ4P6G","created_at":"2026-07-05T09:26:41.713853+00:00"},{"alias_kind":"pith_short_8","alias_value":"V26VZM4J","created_at":"2026-07-05T09:26:41.713853+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.02406","citing_title":"Evaluating AI-Generated Images of Cultural Artifacts with Community-Informed Rubrics","ref_index":111,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02406","citing_title":"Evaluating AI-Generated Images of Cultural Artifacts with Community-Informed Rubrics","ref_index":111,"is_internal_anchor":false},{"citing_arxiv_id":"2412.05579","citing_title":"LLMs-as-Judges: A Comprehensive Survey on LLM-based Evaluation Methods","ref_index":219,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05955","citing_title":"Does Pass Rate Tell the Whole Story? Evaluating Design Constraint Compliance in LLM-based Issue Resolution","ref_index":35,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/V26VZM4JY2DZ4P6GZPWNJQHIMR","json":"https://pith.science/pith/V26VZM4JY2DZ4P6GZPWNJQHIMR.json","graph_json":"https://pith.science/api/pith-number/V26VZM4JY2DZ4P6GZPWNJQHIMR/graph.json","events_json":"https://pith.science/api/pith-number/V26VZM4JY2DZ4P6GZPWNJQHIMR/events.json","paper":"https://pith.science/paper/V26VZM4J"},"agent_actions":{"view_html":"https://pith.science/pith/V26VZM4JY2DZ4P6GZPWNJQHIMR","download_json":"https://pith.science/pith/V26VZM4JY2DZ4P6GZPWNJQHIMR.json","view_paper":"https://pith.science/paper/V26VZM4J","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.20266&json=true","fetch_graph":"https://pith.science/api/pith-number/V26VZM4JY2DZ4P6GZPWNJQHIMR/graph.json","fetch_events":"https://pith.science/api/pith-number/V26VZM4JY2DZ4P6GZPWNJQHIMR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/V26VZM4JY2DZ4P6GZPWNJQHIMR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/V26VZM4JY2DZ4P6GZPWNJQHIMR/action/storage_attestation","attest_author":"https://pith.science/pith/V26VZM4JY2DZ4P6GZPWNJQHIMR/action/author_attestation","sign_citation":"https://pith.science/pith/V26VZM4JY2DZ4P6GZPWNJQHIMR/action/citation_signature","submit_replication":"https://pith.science/pith/V26VZM4JY2DZ4P6GZPWNJQHIMR/action/replication_record"}},"created_at":"2026-07-05T09:26:41.713853+00:00","updated_at":"2026-07-05T09:26:41.713853+00:00"}