{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:K6TJGRV6ZMMYR7OQ2F3IAYB6WV","short_pith_number":"pith:K6TJGRV6","schema_version":"1.0","canonical_sha256":"57a69346becb1988fdd0d17680603eb555d2683537c5594599bf14a2f00a1c9e","source":{"kind":"arxiv","id":"2407.06551","version":2},"attestation_state":"computed","paper":{"title":"OffsetBias: Leveraging Debiased Data for Tuning Evaluators","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Daeyoung Kim, Junsoo Park, Meiying Ren, Sanghyuk Choi, Seungyeon Jwa","submitted_at":"2024-07-09T05:16:22Z","abstract_excerpt":"Employing Large Language Models (LLMs) to assess the quality of generated responses, such as prompting instruct-tuned models or fine-tuning judge models, has become a widely adopted evaluation method. It is also known that such evaluators are vulnerable to biases, such as favoring longer responses. While it is important to overcome this problem, the specifics of these biases remain under-explored. In this work, we qualitatively identify six types of biases inherent in various judge models. We propose EvalBiasBench as a meta-evaluation collection of hand-crafted test cases for each bias type. A"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.06551","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-07-09T05:16:22Z","cross_cats_sorted":[],"title_canon_sha256":"3166afac418cb36461da48a3701db7831f12cf2e4a9a61e596899475cbf61d16","abstract_canon_sha256":"a006a0a7f3462ccb020cf9ed3b775e305f82c1bbd8e50f391865dde0b8000259"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:16:50.935229Z","signature_b64":"emF/P1iOv68ogBwnlIz2B8O32wNYtEMwqBnraY6hqng6T1BQlWhyutVd4Mvmb1l6jf+88GvuhGhtBZafdVaYDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"57a69346becb1988fdd0d17680603eb555d2683537c5594599bf14a2f00a1c9e","last_reissued_at":"2026-07-05T09:16:50.934710Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:16:50.934710Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"OffsetBias: Leveraging Debiased Data for Tuning Evaluators","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Daeyoung Kim, Junsoo Park, Meiying Ren, Sanghyuk Choi, Seungyeon Jwa","submitted_at":"2024-07-09T05:16:22Z","abstract_excerpt":"Employing Large Language Models (LLMs) to assess the quality of generated responses, such as prompting instruct-tuned models or fine-tuning judge models, has become a widely adopted evaluation method. It is also known that such evaluators are vulnerable to biases, such as favoring longer responses. While it is important to overcome this problem, the specifics of these biases remain under-explored. In this work, we qualitatively identify six types of biases inherent in various judge models. We propose EvalBiasBench as a meta-evaluation collection of hand-crafted test cases for each bias type. A"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.06551","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.06551/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.06551","created_at":"2026-07-05T09:16:50.934759+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.06551v2","created_at":"2026-07-05T09:16:50.934759+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.06551","created_at":"2026-07-05T09:16:50.934759+00:00"},{"alias_kind":"pith_short_12","alias_value":"K6TJGRV6ZMMY","created_at":"2026-07-05T09:16:50.934759+00:00"},{"alias_kind":"pith_short_16","alias_value":"K6TJGRV6ZMMYR7OQ","created_at":"2026-07-05T09:16:50.934759+00:00"},{"alias_kind":"pith_short_8","alias_value":"K6TJGRV6","created_at":"2026-07-05T09:16:50.934759+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2411.15594","citing_title":"A Survey on LLM-as-a-Judge","ref_index":113,"is_internal_anchor":false},{"citing_arxiv_id":"2506.01937","citing_title":"RewardBench 2: Advancing Reward Model Evaluation","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2507.06419","citing_title":"Teach a Reward Model to Correct Itself: Reward Guided Adversarial Failure Discovery for Robust Reward Modeling","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2509.23542","citing_title":"On the Shelf Life of Fine-Tuned LLM-Judges: Future-Proofing, Backward-Compatibility, and Question Generalization","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2410.18451","citing_title":"Skywork-Reward: Bag of Tricks for Reward Modeling in LLMs","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03264","citing_title":"SafeScreen: A Safety-First Screening Framework for Personalized Video Retrieval for Vulnerable Users","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2412.05579","citing_title":"LLMs-as-Judges: A Comprehensive Survey on LLM-based Evaluation Methods","ref_index":179,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23178","citing_title":"Judging the Judges: A Systematic Evaluation of Bias Mitigation Strategies in LLM-as-a-Judge Pipelines","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02765","citing_title":"U-Define: Designing User Workflows for Hard and Soft Constraints in LLM-Based Planning","ref_index":88,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/K6TJGRV6ZMMYR7OQ2F3IAYB6WV","json":"https://pith.science/pith/K6TJGRV6ZMMYR7OQ2F3IAYB6WV.json","graph_json":"https://pith.science/api/pith-number/K6TJGRV6ZMMYR7OQ2F3IAYB6WV/graph.json","events_json":"https://pith.science/api/pith-number/K6TJGRV6ZMMYR7OQ2F3IAYB6WV/events.json","paper":"https://pith.science/paper/K6TJGRV6"},"agent_actions":{"view_html":"https://pith.science/pith/K6TJGRV6ZMMYR7OQ2F3IAYB6WV","download_json":"https://pith.science/pith/K6TJGRV6ZMMYR7OQ2F3IAYB6WV.json","view_paper":"https://pith.science/paper/K6TJGRV6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.06551&json=true","fetch_graph":"https://pith.science/api/pith-number/K6TJGRV6ZMMYR7OQ2F3IAYB6WV/graph.json","fetch_events":"https://pith.science/api/pith-number/K6TJGRV6ZMMYR7OQ2F3IAYB6WV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/K6TJGRV6ZMMYR7OQ2F3IAYB6WV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/K6TJGRV6ZMMYR7OQ2F3IAYB6WV/action/storage_attestation","attest_author":"https://pith.science/pith/K6TJGRV6ZMMYR7OQ2F3IAYB6WV/action/author_attestation","sign_citation":"https://pith.science/pith/K6TJGRV6ZMMYR7OQ2F3IAYB6WV/action/citation_signature","submit_replication":"https://pith.science/pith/K6TJGRV6ZMMYR7OQ2F3IAYB6WV/action/replication_record"}},"created_at":"2026-07-05T09:16:50.934759+00:00","updated_at":"2026-07-05T09:16:50.934759+00:00"}