{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:KJI3JCDDOWYU3L6K7XC65FMVAC","short_pith_number":"pith:KJI3JCDD","schema_version":"1.0","canonical_sha256":"5251b4886375b14dafcafdc5ee95950089e0be9c12aaaae63560f06b90db65e9","source":{"kind":"arxiv","id":"2310.01432","version":3},"attestation_state":"computed","paper":{"title":"Split and Merge: Aligning Position Biases in LLM-based Evaluators","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Chaozheng Wang, Cuiyun Gao, Daoyuan Wu, Pingchuan Ma, Shuai Wang, Yang Liu, Zongjie Li","submitted_at":"2023-09-29T14:38:58Z","abstract_excerpt":"Large language models (LLMs) have shown promise as automated evaluators for assessing the quality of answers generated by AI systems. However, these LLM-based evaluators exhibit position bias, or inconsistency, when used to evaluate candidate answers in pairwise comparisons, favoring either the first or second answer regardless of content. To address this limitation, we propose PORTIA, an alignment-based system designed to mimic human comparison strategies to calibrate position bias in a lightweight yet effective manner. Specifically, PORTIA splits the answers into multiple segments, aligns si"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.01432","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-09-29T14:38:58Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"4ea5e4a6bfea8b354416a4ed3ac5261cfb8e2dae9932cf778fdd2df1bf765595","abstract_canon_sha256":"4d0c7d97c05d21fde60ef3646d51ba6add14f1ff085f0077ff099e3d53f61da7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:45:41.320299Z","signature_b64":"QAtEqByaq2hlQIyMIt6pOY0ZeOibaMal32G2S6speHlq9gYjyhimHvrzioj7u0F7b7nUBNJjHHuHMPnNAu/zBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5251b4886375b14dafcafdc5ee95950089e0be9c12aaaae63560f06b90db65e9","last_reissued_at":"2026-07-05T09:45:41.319812Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:45:41.319812Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Split and Merge: Aligning Position Biases in LLM-based Evaluators","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Chaozheng Wang, Cuiyun Gao, Daoyuan Wu, Pingchuan Ma, Shuai Wang, Yang Liu, Zongjie Li","submitted_at":"2023-09-29T14:38:58Z","abstract_excerpt":"Large language models (LLMs) have shown promise as automated evaluators for assessing the quality of answers generated by AI systems. However, these LLM-based evaluators exhibit position bias, or inconsistency, when used to evaluate candidate answers in pairwise comparisons, favoring either the first or second answer regardless of content. To address this limitation, we propose PORTIA, an alignment-based system designed to mimic human comparison strategies to calibrate position bias in a lightweight yet effective manner. Specifically, PORTIA splits the answers into multiple segments, aligns si"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.01432","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.01432/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.01432","created_at":"2026-07-05T09:45:41.319870+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.01432v3","created_at":"2026-07-05T09:45:41.319870+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.01432","created_at":"2026-07-05T09:45:41.319870+00:00"},{"alias_kind":"pith_short_12","alias_value":"KJI3JCDDOWYU","created_at":"2026-07-05T09:45:41.319870+00:00"},{"alias_kind":"pith_short_16","alias_value":"KJI3JCDDOWYU3L6K","created_at":"2026-07-05T09:45:41.319870+00:00"},{"alias_kind":"pith_short_8","alias_value":"KJI3JCDD","created_at":"2026-07-05T09:45:41.319870+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2412.05579","citing_title":"LLMs-as-Judges: A Comprehensive Survey on LLM-based Evaluation Methods","ref_index":139,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KJI3JCDDOWYU3L6K7XC65FMVAC","json":"https://pith.science/pith/KJI3JCDDOWYU3L6K7XC65FMVAC.json","graph_json":"https://pith.science/api/pith-number/KJI3JCDDOWYU3L6K7XC65FMVAC/graph.json","events_json":"https://pith.science/api/pith-number/KJI3JCDDOWYU3L6K7XC65FMVAC/events.json","paper":"https://pith.science/paper/KJI3JCDD"},"agent_actions":{"view_html":"https://pith.science/pith/KJI3JCDDOWYU3L6K7XC65FMVAC","download_json":"https://pith.science/pith/KJI3JCDDOWYU3L6K7XC65FMVAC.json","view_paper":"https://pith.science/paper/KJI3JCDD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.01432&json=true","fetch_graph":"https://pith.science/api/pith-number/KJI3JCDDOWYU3L6K7XC65FMVAC/graph.json","fetch_events":"https://pith.science/api/pith-number/KJI3JCDDOWYU3L6K7XC65FMVAC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KJI3JCDDOWYU3L6K7XC65FMVAC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KJI3JCDDOWYU3L6K7XC65FMVAC/action/storage_attestation","attest_author":"https://pith.science/pith/KJI3JCDDOWYU3L6K7XC65FMVAC/action/author_attestation","sign_citation":"https://pith.science/pith/KJI3JCDDOWYU3L6K7XC65FMVAC/action/citation_signature","submit_replication":"https://pith.science/pith/KJI3JCDDOWYU3L6K7XC65FMVAC/action/replication_record"}},"created_at":"2026-07-05T09:45:41.319870+00:00","updated_at":"2026-07-05T09:45:41.319870+00:00"}