{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:T6UXLZHDGTWI3HI42B2GH6SFHR","short_pith_number":"pith:T6UXLZHD","schema_version":"1.0","canonical_sha256":"9fa975e4e334ec8d9d1cd07463fa453c58b2b926d57a3a7f0834963f82c3e02f","source":{"kind":"arxiv","id":"2310.05657","version":1},"attestation_state":"computed","paper":{"title":"A Closer Look into Automatic Evaluation Using Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Cheng-Han Chiang, Hung-yi Lee","submitted_at":"2023-10-09T12:12:55Z","abstract_excerpt":"Using large language models (LLMs) to evaluate text quality has recently gained popularity. Some prior works explore the idea of using LLMs for evaluation, while they differ in some details of the evaluation process. In this paper, we analyze LLM evaluation (Chiang and Lee, 2023) and G-Eval (Liu et al., 2023), and we discuss how those details in the evaluation process change how well the ratings given by LLMs correlate with human ratings. We find that the auto Chain-of-Thought (CoT) used in G-Eval does not always make G-Eval more aligned with human ratings. We also show that forcing the LLM to"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.05657","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-10-09T12:12:55Z","cross_cats_sorted":[],"title_canon_sha256":"f9bf1957689d01899db0b5532d89b6a8531939a112cc8227701236e3ed4228eb","abstract_canon_sha256":"34a995ca466bd152b5fddf47593e150bdbc6d0c33688a7db1d150676159ce20f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:58:48.165472Z","signature_b64":"HEPKEkIjmLluoHVToa4+iCGgwbxWVvAgsgVZAXmruBtj3euuOa6RM8unCKq9zg9Yt0l0CWcGCBZiqlMuN30rBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9fa975e4e334ec8d9d1cd07463fa453c58b2b926d57a3a7f0834963f82c3e02f","last_reissued_at":"2026-07-05T06:58:48.164986Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:58:48.164986Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Closer Look into Automatic Evaluation Using Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Cheng-Han Chiang, Hung-yi Lee","submitted_at":"2023-10-09T12:12:55Z","abstract_excerpt":"Using large language models (LLMs) to evaluate text quality has recently gained popularity. Some prior works explore the idea of using LLMs for evaluation, while they differ in some details of the evaluation process. In this paper, we analyze LLM evaluation (Chiang and Lee, 2023) and G-Eval (Liu et al., 2023), and we discuss how those details in the evaluation process change how well the ratings given by LLMs correlate with human ratings. We find that the auto Chain-of-Thought (CoT) used in G-Eval does not always make G-Eval more aligned with human ratings. We also show that forcing the LLM to"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.05657","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.05657/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.05657","created_at":"2026-07-05T06:58:48.165047+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.05657v1","created_at":"2026-07-05T06:58:48.165047+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.05657","created_at":"2026-07-05T06:58:48.165047+00:00"},{"alias_kind":"pith_short_12","alias_value":"T6UXLZHDGTWI","created_at":"2026-07-05T06:58:48.165047+00:00"},{"alias_kind":"pith_short_16","alias_value":"T6UXLZHDGTWI3HI4","created_at":"2026-07-05T06:58:48.165047+00:00"},{"alias_kind":"pith_short_8","alias_value":"T6UXLZHD","created_at":"2026-07-05T06:58:48.165047+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2309.03883","citing_title":"DoLa: Decoding by Contrasting Layers Improves Factuality in Large Language Models","ref_index":97,"is_internal_anchor":false},{"citing_arxiv_id":"2603.10477","citing_title":"PEEM: Prompt Engineering Evaluation Metrics for Interpretable Joint Evaluation of Prompts and Responses","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2412.05579","citing_title":"LLMs-as-Judges: A Comprehensive Survey on LLM-based Evaluation Methods","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06201","citing_title":"Towards Annotation-Free Validation of MLLMs: A Vision-Language Logical Consistency Metric","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/T6UXLZHDGTWI3HI42B2GH6SFHR","json":"https://pith.science/pith/T6UXLZHDGTWI3HI42B2GH6SFHR.json","graph_json":"https://pith.science/api/pith-number/T6UXLZHDGTWI3HI42B2GH6SFHR/graph.json","events_json":"https://pith.science/api/pith-number/T6UXLZHDGTWI3HI42B2GH6SFHR/events.json","paper":"https://pith.science/paper/T6UXLZHD"},"agent_actions":{"view_html":"https://pith.science/pith/T6UXLZHDGTWI3HI42B2GH6SFHR","download_json":"https://pith.science/pith/T6UXLZHDGTWI3HI42B2GH6SFHR.json","view_paper":"https://pith.science/paper/T6UXLZHD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.05657&json=true","fetch_graph":"https://pith.science/api/pith-number/T6UXLZHDGTWI3HI42B2GH6SFHR/graph.json","fetch_events":"https://pith.science/api/pith-number/T6UXLZHDGTWI3HI42B2GH6SFHR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/T6UXLZHDGTWI3HI42B2GH6SFHR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/T6UXLZHDGTWI3HI42B2GH6SFHR/action/storage_attestation","attest_author":"https://pith.science/pith/T6UXLZHDGTWI3HI42B2GH6SFHR/action/author_attestation","sign_citation":"https://pith.science/pith/T6UXLZHDGTWI3HI42B2GH6SFHR/action/citation_signature","submit_replication":"https://pith.science/pith/T6UXLZHDGTWI3HI42B2GH6SFHR/action/replication_record"}},"created_at":"2026-07-05T06:58:48.165047+00:00","updated_at":"2026-07-05T06:58:48.165047+00:00"}