{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:YFKQINBL6766OAZCNUB3UHORLE","short_pith_number":"pith:YFKQINBL","schema_version":"1.0","canonical_sha256":"c15504342bf7fde703226d03ba1dd15901fff1b0d440f44a497dec68684ce015","source":{"kind":"arxiv","id":"2305.01937","version":1},"attestation_state":"computed","paper":{"title":"Can Large Language Models Be an Alternative to Human Evaluations?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.HC"],"primary_cat":"cs.CL","authors_text":"Cheng-Han Chiang, Hung-yi Lee","submitted_at":"2023-05-03T07:28:50Z","abstract_excerpt":"Human evaluation is indispensable and inevitable for assessing the quality of texts generated by machine learning models or written by humans. However, human evaluation is very difficult to reproduce and its quality is notoriously unstable, hindering fair comparisons among different natural language processing (NLP) models and algorithms. Recently, large language models (LLMs) have demonstrated exceptional performance on unseen tasks when only the task instructions are provided. In this paper, we explore if such an ability of the LLMs can be used as an alternative to human evaluation. We prese"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.01937","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-05-03T07:28:50Z","cross_cats_sorted":["cs.HC"],"title_canon_sha256":"ff53edb56af6081336386260d993753caa9f25c3a22676bf4f576f397dff7a05","abstract_canon_sha256":"559411a704d265404e2ad0cf128a04f7d4d764b9ec160dd61215b904c64e5507"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:06:45.720835Z","signature_b64":"maG3BeeagTM8pv/CK/tSycvmvIXbddkcR2iyLobMzVJVAXl9ocezhlBmm92v0VMvHcBjIDOm0H7QivwtZciGAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c15504342bf7fde703226d03ba1dd15901fff1b0d440f44a497dec68684ce015","last_reissued_at":"2026-07-05T06:06:45.720355Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:06:45.720355Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Can Large Language Models Be an Alternative to Human Evaluations?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.HC"],"primary_cat":"cs.CL","authors_text":"Cheng-Han Chiang, Hung-yi Lee","submitted_at":"2023-05-03T07:28:50Z","abstract_excerpt":"Human evaluation is indispensable and inevitable for assessing the quality of texts generated by machine learning models or written by humans. However, human evaluation is very difficult to reproduce and its quality is notoriously unstable, hindering fair comparisons among different natural language processing (NLP) models and algorithms. Recently, large language models (LLMs) have demonstrated exceptional performance on unseen tasks when only the task instructions are provided. In this paper, we explore if such an ability of the LLMs can be used as an alternative to human evaluation. We prese"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.01937","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.01937/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.01937","created_at":"2026-07-05T06:06:45.720417+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.01937v1","created_at":"2026-07-05T06:06:45.720417+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.01937","created_at":"2026-07-05T06:06:45.720417+00:00"},{"alias_kind":"pith_short_12","alias_value":"YFKQINBL6766","created_at":"2026-07-05T06:06:45.720417+00:00"},{"alias_kind":"pith_short_16","alias_value":"YFKQINBL6766OAZC","created_at":"2026-07-05T06:06:45.720417+00:00"},{"alias_kind":"pith_short_8","alias_value":"YFKQINBL","created_at":"2026-07-05T06:06:45.720417+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08403","citing_title":"Game Theory Driven Multi-Agent Framework Mitigates Language Model Hallucination","ref_index":19,"is_internal_anchor":true},{"citing_arxiv_id":"2606.12924","citing_title":"Iterating Toward Better Search: A Two-Agent Simulation Framework for Evaluating Agentic Search Architectures in E-Commerce","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06462","citing_title":"Benchmark Everything Everywhere All at Once","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12068","citing_title":"StanceNakba Shared Task: Actor and Topic-Aware Stance Detection in Public Discourse","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2311.07911","citing_title":"Instruction-Following Evaluation for Large Language Models","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2401.02458","citing_title":"Data-Centric Foundation Models in Computational Healthcare: A Survey","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15416","citing_title":"Margin-Adaptive Confidence Ranking for Reliable LLM Judgement","ref_index":117,"is_internal_anchor":false},{"citing_arxiv_id":"2309.03883","citing_title":"DoLa: Decoding by Contrasting Layers Improves Factuality in Large Language Models","ref_index":95,"is_internal_anchor":false},{"citing_arxiv_id":"2311.16867","citing_title":"The Falcon Series of Open Language Models","ref_index":260,"is_internal_anchor":false},{"citing_arxiv_id":"2311.05232","citing_title":"A Survey on Hallucination in Large Language Models: Principles, Taxonomy, Challenges, and Open Questions","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2604.28049","citing_title":"Agent-Agnostic Evaluation of SQL Accuracy in Production Text-to-SQL Systems","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06283","citing_title":"Quantifying the Statistical Effect of Rubric Modifications on Human-Autorater Agreement","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2306.05685","citing_title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17114","citing_title":"The Provenance Gap in Clinical AI: Evidence-Traceable Temporal Knowledge Graphs for Rare Disease Reasoning","ref_index":42,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YFKQINBL6766OAZCNUB3UHORLE","json":"https://pith.science/pith/YFKQINBL6766OAZCNUB3UHORLE.json","graph_json":"https://pith.science/api/pith-number/YFKQINBL6766OAZCNUB3UHORLE/graph.json","events_json":"https://pith.science/api/pith-number/YFKQINBL6766OAZCNUB3UHORLE/events.json","paper":"https://pith.science/paper/YFKQINBL"},"agent_actions":{"view_html":"https://pith.science/pith/YFKQINBL6766OAZCNUB3UHORLE","download_json":"https://pith.science/pith/YFKQINBL6766OAZCNUB3UHORLE.json","view_paper":"https://pith.science/paper/YFKQINBL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.01937&json=true","fetch_graph":"https://pith.science/api/pith-number/YFKQINBL6766OAZCNUB3UHORLE/graph.json","fetch_events":"https://pith.science/api/pith-number/YFKQINBL6766OAZCNUB3UHORLE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YFKQINBL6766OAZCNUB3UHORLE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YFKQINBL6766OAZCNUB3UHORLE/action/storage_attestation","attest_author":"https://pith.science/pith/YFKQINBL6766OAZCNUB3UHORLE/action/author_attestation","sign_citation":"https://pith.science/pith/YFKQINBL6766OAZCNUB3UHORLE/action/citation_signature","submit_replication":"https://pith.science/pith/YFKQINBL6766OAZCNUB3UHORLE/action/replication_record"}},"created_at":"2026-07-05T06:06:45.720417+00:00","updated_at":"2026-07-05T06:06:45.720417+00:00"}