{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:QOBJOS5GBFWTFRNHMPKTSWZJND","short_pith_number":"pith:QOBJOS5G","schema_version":"1.0","canonical_sha256":"8382974ba6096d32c5a763d5395b2968e67e4f3809ba07d266c8f6c246ee89c5","source":{"kind":"arxiv","id":"2505.16003","version":1},"attestation_state":"computed","paper":{"title":"SLMEval: Entropy-Based Calibration for Human-Aligned Evaluation of Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Christopher Clarke, Jason Mars, Krisztian Flautner, Lingjia Tang, Roland Daynauth","submitted_at":"2025-05-21T20:40:30Z","abstract_excerpt":"The LLM-as-a-Judge paradigm offers a scalable, reference-free approach for evaluating language models. Although several calibration techniques have been proposed to better align these evaluators with human judgment, prior studies focus primarily on narrow, well-structured benchmarks. As a result, it remains unclear whether such calibrations generalize to real-world, open-ended tasks.\n  In this work, we show that SOTA calibrated evaluators often fail in these settings, exhibiting weak or even negative correlation with human judgments. To address this, we propose SLMEval, a novel and efficient c"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.16003","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-05-21T20:40:30Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"f9d026ed5e3eda66bca9428bdcf6d2dfe43f829ac3eb4f2f271c19cd892200e8","abstract_canon_sha256":"a746c6c182e9a293c5635847e6391968e1b8138f01efbfafad765b0d7365480f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:07:28.457367Z","signature_b64":"SpY3JCDm925Od6GUhwwj4a0CKcAkMonOheNhOEBQPOu4YcyXyXS15nnOHTHq//ZVgq5Wdrioz1mLLSoq++yRBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8382974ba6096d32c5a763d5395b2968e67e4f3809ba07d266c8f6c246ee89c5","last_reissued_at":"2026-07-05T11:07:28.456837Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:07:28.456837Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SLMEval: Entropy-Based Calibration for Human-Aligned Evaluation of Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Christopher Clarke, Jason Mars, Krisztian Flautner, Lingjia Tang, Roland Daynauth","submitted_at":"2025-05-21T20:40:30Z","abstract_excerpt":"The LLM-as-a-Judge paradigm offers a scalable, reference-free approach for evaluating language models. Although several calibration techniques have been proposed to better align these evaluators with human judgment, prior studies focus primarily on narrow, well-structured benchmarks. As a result, it remains unclear whether such calibrations generalize to real-world, open-ended tasks.\n  In this work, we show that SOTA calibrated evaluators often fail in these settings, exhibiting weak or even negative correlation with human judgments. To address this, we propose SLMEval, a novel and efficient c"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.16003","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.16003/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.16003","created_at":"2026-07-05T11:07:28.456891+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.16003v1","created_at":"2026-07-05T11:07:28.456891+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.16003","created_at":"2026-07-05T11:07:28.456891+00:00"},{"alias_kind":"pith_short_12","alias_value":"QOBJOS5GBFWT","created_at":"2026-07-05T11:07:28.456891+00:00"},{"alias_kind":"pith_short_16","alias_value":"QOBJOS5GBFWTFRNH","created_at":"2026-07-05T11:07:28.456891+00:00"},{"alias_kind":"pith_short_8","alias_value":"QOBJOS5G","created_at":"2026-07-05T11:07:28.456891+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QOBJOS5GBFWTFRNHMPKTSWZJND","json":"https://pith.science/pith/QOBJOS5GBFWTFRNHMPKTSWZJND.json","graph_json":"https://pith.science/api/pith-number/QOBJOS5GBFWTFRNHMPKTSWZJND/graph.json","events_json":"https://pith.science/api/pith-number/QOBJOS5GBFWTFRNHMPKTSWZJND/events.json","paper":"https://pith.science/paper/QOBJOS5G"},"agent_actions":{"view_html":"https://pith.science/pith/QOBJOS5GBFWTFRNHMPKTSWZJND","download_json":"https://pith.science/pith/QOBJOS5GBFWTFRNHMPKTSWZJND.json","view_paper":"https://pith.science/paper/QOBJOS5G","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.16003&json=true","fetch_graph":"https://pith.science/api/pith-number/QOBJOS5GBFWTFRNHMPKTSWZJND/graph.json","fetch_events":"https://pith.science/api/pith-number/QOBJOS5GBFWTFRNHMPKTSWZJND/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QOBJOS5GBFWTFRNHMPKTSWZJND/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QOBJOS5GBFWTFRNHMPKTSWZJND/action/storage_attestation","attest_author":"https://pith.science/pith/QOBJOS5GBFWTFRNHMPKTSWZJND/action/author_attestation","sign_citation":"https://pith.science/pith/QOBJOS5GBFWTFRNHMPKTSWZJND/action/citation_signature","submit_replication":"https://pith.science/pith/QOBJOS5GBFWTFRNHMPKTSWZJND/action/replication_record"}},"created_at":"2026-07-05T11:07:28.456891+00:00","updated_at":"2026-07-05T11:07:28.456891+00:00"}