{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:C4APT4Y64Y3BVCCN35GTD47DM3","short_pith_number":"pith:C4APT4Y6","schema_version":"1.0","canonical_sha256":"1700f9f31ee6361a884ddf4d31f3e366ddf498f3d32a268ed656216d04cd28fc","source":{"kind":"arxiv","id":"2402.10770","version":4},"attestation_state":"computed","paper":{"title":"How Reliable Are Automatic Evaluation Methods for Instruction-Tuned LLMs?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Ehsan Doostmohammadi, Marco Kuhlmann, Oskar Holmstr\\\"om","submitted_at":"2024-02-16T15:48:33Z","abstract_excerpt":"Work on instruction-tuned Large Language Models (LLMs) has used automatic methods based on text overlap and LLM judgments as cost-effective alternatives to human evaluation. In this paper, we perform a meta-evaluation of such methods and assess their reliability across a broad range of tasks. In evaluating how well automatic methods align with human evaluations, correlation metrics are the most commonly employed method despite their inherent limitations when dealing with ties and different scales. To address these shortcomings, we use Pairwise Accuracy as an alternative to standard correlation"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.10770","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-02-16T15:48:33Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"d1f2c397423c880c8ab9e9b7b8f7769b8923042037578f6cc7abdf2e267857b4","abstract_canon_sha256":"670b435fdf876ccb99c9cbdbd5993ad0c0d5375302e5dec8e64ba0160609d16f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:14:30.330589Z","signature_b64":"3zAS4ncmiBJe3qTBUKPH2BsgHsbXmU2aK0UTKzqDIMD2Rd2UwoMwP8ZuGcRu9F1eAdK9N0wcMH/YOyBV/B4BBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1700f9f31ee6361a884ddf4d31f3e366ddf498f3d32a268ed656216d04cd28fc","last_reissued_at":"2026-07-05T09:14:30.329980Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:14:30.329980Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"How Reliable Are Automatic Evaluation Methods for Instruction-Tuned LLMs?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Ehsan Doostmohammadi, Marco Kuhlmann, Oskar Holmstr\\\"om","submitted_at":"2024-02-16T15:48:33Z","abstract_excerpt":"Work on instruction-tuned Large Language Models (LLMs) has used automatic methods based on text overlap and LLM judgments as cost-effective alternatives to human evaluation. In this paper, we perform a meta-evaluation of such methods and assess their reliability across a broad range of tasks. In evaluating how well automatic methods align with human evaluations, correlation metrics are the most commonly employed method despite their inherent limitations when dealing with ties and different scales. To address these shortcomings, we use Pairwise Accuracy as an alternative to standard correlation"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.10770","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.10770/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.10770","created_at":"2026-07-05T09:14:30.330041+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.10770v4","created_at":"2026-07-05T09:14:30.330041+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.10770","created_at":"2026-07-05T09:14:30.330041+00:00"},{"alias_kind":"pith_short_12","alias_value":"C4APT4Y64Y3B","created_at":"2026-07-05T09:14:30.330041+00:00"},{"alias_kind":"pith_short_16","alias_value":"C4APT4Y64Y3BVCCN","created_at":"2026-07-05T09:14:30.330041+00:00"},{"alias_kind":"pith_short_8","alias_value":"C4APT4Y6","created_at":"2026-07-05T09:14:30.330041+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.01333","citing_title":"Can Large Language Models Serve as Evaluators for Code Summarization?","ref_index":63,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/C4APT4Y64Y3BVCCN35GTD47DM3","json":"https://pith.science/pith/C4APT4Y64Y3BVCCN35GTD47DM3.json","graph_json":"https://pith.science/api/pith-number/C4APT4Y64Y3BVCCN35GTD47DM3/graph.json","events_json":"https://pith.science/api/pith-number/C4APT4Y64Y3BVCCN35GTD47DM3/events.json","paper":"https://pith.science/paper/C4APT4Y6"},"agent_actions":{"view_html":"https://pith.science/pith/C4APT4Y64Y3BVCCN35GTD47DM3","download_json":"https://pith.science/pith/C4APT4Y64Y3BVCCN35GTD47DM3.json","view_paper":"https://pith.science/paper/C4APT4Y6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.10770&json=true","fetch_graph":"https://pith.science/api/pith-number/C4APT4Y64Y3BVCCN35GTD47DM3/graph.json","fetch_events":"https://pith.science/api/pith-number/C4APT4Y64Y3BVCCN35GTD47DM3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/C4APT4Y64Y3BVCCN35GTD47DM3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/C4APT4Y64Y3BVCCN35GTD47DM3/action/storage_attestation","attest_author":"https://pith.science/pith/C4APT4Y64Y3BVCCN35GTD47DM3/action/author_attestation","sign_citation":"https://pith.science/pith/C4APT4Y64Y3BVCCN35GTD47DM3/action/citation_signature","submit_replication":"https://pith.science/pith/C4APT4Y64Y3BVCCN35GTD47DM3/action/replication_record"}},"created_at":"2026-07-05T09:14:30.330041+00:00","updated_at":"2026-07-05T09:14:30.330041+00:00"}