{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:HCF4GHYCK7IGU3PTS42ZTEVQY3","short_pith_number":"pith:HCF4GHYC","schema_version":"1.0","canonical_sha256":"388bc31f0257d06a6df397359992b0c6cbcf6bd2658e5c11589aa9fdedf40ab2","source":{"kind":"arxiv","id":"2410.15393","version":1},"attestation_state":"computed","paper":{"title":"CalibraEval: Calibrating Prediction Distribution to Mitigate Selection Bias in LLMs-as-Judges","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Haitao Li, Junjie Chen, Qian Dong, Qingyao Ai, Yiqun Liu, Yujia Zhou, Zhumin Chu","submitted_at":"2024-10-20T13:47:39Z","abstract_excerpt":"The use of large language models (LLMs) as automated evaluation tools to assess the quality of generated natural language, known as LLMs-as-Judges, has demonstrated promising capabilities and is rapidly gaining widespread attention. However, when applied to pairwise comparisons of candidate responses, LLM-based evaluators often exhibit selection bias. Specifically, their judgments may become inconsistent when the option positions or ID tokens are swapped, compromising the effectiveness and fairness of the evaluation result. To address this challenge, we introduce CalibraEval, a novel label-fre"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.15393","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-10-20T13:47:39Z","cross_cats_sorted":[],"title_canon_sha256":"a34512b51c76c6947a08a160ba8a517e1cb658442ffe91b9775199d191cdd08a","abstract_canon_sha256":"87afc614f8fd1ccef7cd3b4705f1a763a5c683a79f7f0d7bff8e75d1e29d3620"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:23:18.856506Z","signature_b64":"OxWqEVC6K6lJPhWS+FztInb3FZmQvnIQ4W3KpugaLz4XBwNNw5f/Dy31xkPJdxuN9sCUhQcIxDlhY9UZNifiAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"388bc31f0257d06a6df397359992b0c6cbcf6bd2658e5c11589aa9fdedf40ab2","last_reissued_at":"2026-07-05T09:23:18.856070Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:23:18.856070Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CalibraEval: Calibrating Prediction Distribution to Mitigate Selection Bias in LLMs-as-Judges","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Haitao Li, Junjie Chen, Qian Dong, Qingyao Ai, Yiqun Liu, Yujia Zhou, Zhumin Chu","submitted_at":"2024-10-20T13:47:39Z","abstract_excerpt":"The use of large language models (LLMs) as automated evaluation tools to assess the quality of generated natural language, known as LLMs-as-Judges, has demonstrated promising capabilities and is rapidly gaining widespread attention. However, when applied to pairwise comparisons of candidate responses, LLM-based evaluators often exhibit selection bias. Specifically, their judgments may become inconsistent when the option positions or ID tokens are swapped, compromising the effectiveness and fairness of the evaluation result. To address this challenge, we introduce CalibraEval, a novel label-fre"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.15393","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.15393/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.15393","created_at":"2026-07-05T09:23:18.856133+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.15393v1","created_at":"2026-07-05T09:23:18.856133+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.15393","created_at":"2026-07-05T09:23:18.856133+00:00"},{"alias_kind":"pith_short_12","alias_value":"HCF4GHYCK7IG","created_at":"2026-07-05T09:23:18.856133+00:00"},{"alias_kind":"pith_short_16","alias_value":"HCF4GHYCK7IGU3PT","created_at":"2026-07-05T09:23:18.856133+00:00"},{"alias_kind":"pith_short_8","alias_value":"HCF4GHYC","created_at":"2026-07-05T09:23:18.856133+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07774","citing_title":"ScopeJudge: Cost-Aware Pre-Execution Gating for Offensive Security Agents","ref_index":19,"is_internal_anchor":true},{"citing_arxiv_id":"2605.10528","citing_title":"Collective Alignment in LLM Multi-Agent Systems: Disentangling Bias from Cooperation via Statistical Physics","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2412.05579","citing_title":"LLMs-as-Judges: A Comprehensive Survey on LLM-based Evaluation Methods","ref_index":132,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HCF4GHYCK7IGU3PTS42ZTEVQY3","json":"https://pith.science/pith/HCF4GHYCK7IGU3PTS42ZTEVQY3.json","graph_json":"https://pith.science/api/pith-number/HCF4GHYCK7IGU3PTS42ZTEVQY3/graph.json","events_json":"https://pith.science/api/pith-number/HCF4GHYCK7IGU3PTS42ZTEVQY3/events.json","paper":"https://pith.science/paper/HCF4GHYC"},"agent_actions":{"view_html":"https://pith.science/pith/HCF4GHYCK7IGU3PTS42ZTEVQY3","download_json":"https://pith.science/pith/HCF4GHYCK7IGU3PTS42ZTEVQY3.json","view_paper":"https://pith.science/paper/HCF4GHYC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.15393&json=true","fetch_graph":"https://pith.science/api/pith-number/HCF4GHYCK7IGU3PTS42ZTEVQY3/graph.json","fetch_events":"https://pith.science/api/pith-number/HCF4GHYCK7IGU3PTS42ZTEVQY3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HCF4GHYCK7IGU3PTS42ZTEVQY3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HCF4GHYCK7IGU3PTS42ZTEVQY3/action/storage_attestation","attest_author":"https://pith.science/pith/HCF4GHYCK7IGU3PTS42ZTEVQY3/action/author_attestation","sign_citation":"https://pith.science/pith/HCF4GHYCK7IGU3PTS42ZTEVQY3/action/citation_signature","submit_replication":"https://pith.science/pith/HCF4GHYCK7IGU3PTS42ZTEVQY3/action/replication_record"}},"created_at":"2026-07-05T09:23:18.856133+00:00","updated_at":"2026-07-05T09:23:18.856133+00:00"}