{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:VSWSIS5G2RZKD7QP2DUR6VLQSM","short_pith_number":"pith:VSWSIS5G","schema_version":"1.0","canonical_sha256":"acad244ba6d472a1fe0fd0e91f55709311c6aec69f0e8fb5af15a92dce96082a","source":{"kind":"arxiv","id":"2403.16950","version":5},"attestation_state":"computed","paper":{"title":"Aligning with Human Judgement: The Role of Pairwise Preference in Large Language Model Evaluators","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Anna Korhonen, Ehsan Shareghi, Han Zhou, Ivan Vuli\\'c, Nigel Collier, Yinhong Liu, Zhijiang Guo","submitted_at":"2024-03-25T17:11:28Z","abstract_excerpt":"Large Language Models (LLMs) have demonstrated promising capabilities as automatic evaluators in assessing the quality of generated natural language. However, LLMs still exhibit biases in evaluation and often struggle to generate coherent evaluations that align with human assessments. In this work, we first conduct a systematic study of the misalignment between LLM evaluators and human evaluation, revealing that existing calibration methods aimed at mitigating biases of LLMs are insufficient for effectively aligning LLM evaluators. Inspired by the use of preference data in RLHF, we formulate t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.16950","kind":"arxiv","version":5},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-03-25T17:11:28Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"626292c20883a5a3427ff6d0c38731951076738aae74fac6bca944828a66ad58","abstract_canon_sha256":"beebf2153b2871d95f97c4b4ea64478ef73b526559130df4e7f42fc0904403d4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:02:02.834431Z","signature_b64":"tZw/3H6RcKYm5aUbcW2o5yOz4EDXZgMHHkwbobft/urMz8Y7jUqNrUdbwxp5kPUvnirySCyPxswBdZdatcaICw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"acad244ba6d472a1fe0fd0e91f55709311c6aec69f0e8fb5af15a92dce96082a","last_reissued_at":"2026-07-05T10:02:02.833997Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:02:02.833997Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Aligning with Human Judgement: The Role of Pairwise Preference in Large Language Model Evaluators","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Anna Korhonen, Ehsan Shareghi, Han Zhou, Ivan Vuli\\'c, Nigel Collier, Yinhong Liu, Zhijiang Guo","submitted_at":"2024-03-25T17:11:28Z","abstract_excerpt":"Large Language Models (LLMs) have demonstrated promising capabilities as automatic evaluators in assessing the quality of generated natural language. However, LLMs still exhibit biases in evaluation and often struggle to generate coherent evaluations that align with human assessments. In this work, we first conduct a systematic study of the misalignment between LLM evaluators and human evaluation, revealing that existing calibration methods aimed at mitigating biases of LLMs are insufficient for effectively aligning LLM evaluators. Inspired by the use of preference data in RLHF, we formulate t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.16950","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.16950/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.16950","created_at":"2026-07-05T10:02:02.834053+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.16950v5","created_at":"2026-07-05T10:02:02.834053+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.16950","created_at":"2026-07-05T10:02:02.834053+00:00"},{"alias_kind":"pith_short_12","alias_value":"VSWSIS5G2RZK","created_at":"2026-07-05T10:02:02.834053+00:00"},{"alias_kind":"pith_short_16","alias_value":"VSWSIS5G2RZKD7QP","created_at":"2026-07-05T10:02:02.834053+00:00"},{"alias_kind":"pith_short_8","alias_value":"VSWSIS5G","created_at":"2026-07-05T10:02:02.834053+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24004","citing_title":"Towards Spec Learning: Inference-Time Alignment from Preference Pairs","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05384","citing_title":"Stability vs. Manipulability: Evaluating Robustness Under Post-Decision Interaction in LLM Judges","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2606.24004","citing_title":"Towards Spec Learning: Inference-Time Alignment from Preference Pairs","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27446","citing_title":"Causal Connections: Leveraging Multilingual Fine-Tuning for Financial QA@FinCausal 2026","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27316","citing_title":"LLM-Based Examination of Eligibility Criteria from Securities Prospectuses at the German Central Bank","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19141","citing_title":"GRASP: Deterministic argument ranking in interaction graphs","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2506.14092","citing_title":"Fragile Preferences: A Deep Dive Into Order Effects in Large Language Models","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16304","citing_title":"Results-Actionability Gap: Understanding How Practitioners Evaluate LLM Products in the Wild","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02655","citing_title":"Semantic Data Processing with Holistic Data Understanding","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2412.05579","citing_title":"LLMs-as-Judges: A Comprehensive Survey on LLM-based Evaluation Methods","ref_index":154,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23178","citing_title":"Judging the Judges: A Systematic Evaluation of Bias Mitigation Strategies in LLM-as-a-Judge Pipelines","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VSWSIS5G2RZKD7QP2DUR6VLQSM","json":"https://pith.science/pith/VSWSIS5G2RZKD7QP2DUR6VLQSM.json","graph_json":"https://pith.science/api/pith-number/VSWSIS5G2RZKD7QP2DUR6VLQSM/graph.json","events_json":"https://pith.science/api/pith-number/VSWSIS5G2RZKD7QP2DUR6VLQSM/events.json","paper":"https://pith.science/paper/VSWSIS5G"},"agent_actions":{"view_html":"https://pith.science/pith/VSWSIS5G2RZKD7QP2DUR6VLQSM","download_json":"https://pith.science/pith/VSWSIS5G2RZKD7QP2DUR6VLQSM.json","view_paper":"https://pith.science/paper/VSWSIS5G","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.16950&json=true","fetch_graph":"https://pith.science/api/pith-number/VSWSIS5G2RZKD7QP2DUR6VLQSM/graph.json","fetch_events":"https://pith.science/api/pith-number/VSWSIS5G2RZKD7QP2DUR6VLQSM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VSWSIS5G2RZKD7QP2DUR6VLQSM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VSWSIS5G2RZKD7QP2DUR6VLQSM/action/storage_attestation","attest_author":"https://pith.science/pith/VSWSIS5G2RZKD7QP2DUR6VLQSM/action/author_attestation","sign_citation":"https://pith.science/pith/VSWSIS5G2RZKD7QP2DUR6VLQSM/action/citation_signature","submit_replication":"https://pith.science/pith/VSWSIS5G2RZKD7QP2DUR6VLQSM/action/replication_record"}},"created_at":"2026-07-05T10:02:02.834053+00:00","updated_at":"2026-07-05T10:02:02.834053+00:00"}