{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:73IB5S7GCCAUSSMRV2NEG6A7TR","short_pith_number":"pith:73IB5S7G","schema_version":"1.0","canonical_sha256":"fed01ecbe61081494991ae9a43781f9c4a49ab3052494b6d56d31effe0e5cdd1","source":{"kind":"arxiv","id":"2402.10669","version":5},"attestation_state":"computed","paper":{"title":"Humans or LLMs as the Judge? A Study on Judgement Biases","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Benyou Wang, Feng Jiang, Guiming Hardy Chen, Shunian Chen, Ziche Liu","submitted_at":"2024-02-16T13:21:06Z","abstract_excerpt":"Adopting human and large language models (LLM) as judges (a.k.a human- and LLM-as-a-judge) for evaluating the performance of LLMs has recently gained attention. Nonetheless, this approach concurrently introduces potential biases from human and LLMs, questioning the reliability of the evaluation results. In this paper, we propose a novel framework that is free from referencing groundtruth annotations for investigating Misinformation Oversight Bias, Gender Bias, Authority Bias and Beauty Bias on LLM and human judges. We curate a dataset referring to the revised Bloom's Taxonomy and conduct thous"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.10669","kind":"arxiv","version":5},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-02-16T13:21:06Z","cross_cats_sorted":[],"title_canon_sha256":"c9a18699c90237d94ba48b9d55eb4552ef32c94dd12b9203626d671b79b00ab9","abstract_canon_sha256":"648b583f26c56543c2c5f69dad859dbd122846823dfbc5b4c73449019cf91e5e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:12:02.655717Z","signature_b64":"qyHVntHeO1eNw108IetlJl22I0qunaIkA3cNzHr4JBMj6sGd6RHh0WL/xK/aDLxHvmaWwz0mZU/U2i/697kPDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fed01ecbe61081494991ae9a43781f9c4a49ab3052494b6d56d31effe0e5cdd1","last_reissued_at":"2026-07-05T09:12:02.655309Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:12:02.655309Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Humans or LLMs as the Judge? A Study on Judgement Biases","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Benyou Wang, Feng Jiang, Guiming Hardy Chen, Shunian Chen, Ziche Liu","submitted_at":"2024-02-16T13:21:06Z","abstract_excerpt":"Adopting human and large language models (LLM) as judges (a.k.a human- and LLM-as-a-judge) for evaluating the performance of LLMs has recently gained attention. Nonetheless, this approach concurrently introduces potential biases from human and LLMs, questioning the reliability of the evaluation results. In this paper, we propose a novel framework that is free from referencing groundtruth annotations for investigating Misinformation Oversight Bias, Gender Bias, Authority Bias and Beauty Bias on LLM and human judges. We curate a dataset referring to the revised Bloom's Taxonomy and conduct thous"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.10669","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.10669/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.10669","created_at":"2026-07-05T09:12:02.655381+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.10669v5","created_at":"2026-07-05T09:12:02.655381+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.10669","created_at":"2026-07-05T09:12:02.655381+00:00"},{"alias_kind":"pith_short_12","alias_value":"73IB5S7GCCAU","created_at":"2026-07-05T09:12:02.655381+00:00"},{"alias_kind":"pith_short_16","alias_value":"73IB5S7GCCAUSSMR","created_at":"2026-07-05T09:12:02.655381+00:00"},{"alias_kind":"pith_short_8","alias_value":"73IB5S7G","created_at":"2026-07-05T09:12:02.655381+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":18,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24428","citing_title":"Escaping the Self-Confirmation Trap: An Execute-Distill-Verify Paradigm for Agentic Experience Learning","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11635","citing_title":"Are LLMs Bad at Moral Reasoning?","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12422","citing_title":"Creating and Evaluating K-12 GenAI Assessment Graders Through Context Engineering","ref_index":236,"is_internal_anchor":false},{"citing_arxiv_id":"2410.20791","citing_title":"From Cool Demos to Production-Ready FMware: Core Challenges and a Technology Roadmap","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2411.15594","citing_title":"A Survey on LLM-as-a-Judge","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18805","citing_title":"RecoAtlas: From Semantic Plausibility to Set-Level Utility in LLM Recommendation Agents","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2407.21772","citing_title":"ShieldGemma: Generative AI Content Moderation Based on Gemma","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2507.22359","citing_title":"League of LLMs: A Benchmark-Free Paradigm for Mutual Evaluation of Large Language Models","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2511.20284","citing_title":"Can LLMs Make (Personalized) Access Control Decisions?","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2405.14782","citing_title":"Lessons from the Trenches on Reproducible Evaluation of Language Models","ref_index":255,"is_internal_anchor":false},{"citing_arxiv_id":"2603.03332","citing_title":"Fragile Thoughts: How Large Language Models Handle Chain-of-Thought Perturbations","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02359","citing_title":"Using LLM-as-a-Judge/Jury to Advance Scalable, Clinically-Validated Safety Evaluations of Model Responses to Users Demonstrating Psychosis","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27132","citing_title":"TRUST: A Framework for Decentralized AI Service v.0.1","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2412.05579","citing_title":"LLMs-as-Judges: A Comprehensive Survey on LLM-based Evaluation Methods","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23178","citing_title":"Judging the Judges: A Systematic Evaluation of Bias Mitigation Strategies in LLM-as-a-Judge Pipelines","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21769","citing_title":"Who Defines \"Best\"? Towards Interactive, User-Defined Evaluation of LLM Leaderboards","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18835","citing_title":"Semantic Needles in Document Haystacks: Sensitivity Testing of LLM-as-a-Judge Similarity Scoring","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09791","citing_title":"Pioneer Agent: Continual Improvement of Small Language Models in Production","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/73IB5S7GCCAUSSMRV2NEG6A7TR","json":"https://pith.science/pith/73IB5S7GCCAUSSMRV2NEG6A7TR.json","graph_json":"https://pith.science/api/pith-number/73IB5S7GCCAUSSMRV2NEG6A7TR/graph.json","events_json":"https://pith.science/api/pith-number/73IB5S7GCCAUSSMRV2NEG6A7TR/events.json","paper":"https://pith.science/paper/73IB5S7G"},"agent_actions":{"view_html":"https://pith.science/pith/73IB5S7GCCAUSSMRV2NEG6A7TR","download_json":"https://pith.science/pith/73IB5S7GCCAUSSMRV2NEG6A7TR.json","view_paper":"https://pith.science/paper/73IB5S7G","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.10669&json=true","fetch_graph":"https://pith.science/api/pith-number/73IB5S7GCCAUSSMRV2NEG6A7TR/graph.json","fetch_events":"https://pith.science/api/pith-number/73IB5S7GCCAUSSMRV2NEG6A7TR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/73IB5S7GCCAUSSMRV2NEG6A7TR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/73IB5S7GCCAUSSMRV2NEG6A7TR/action/storage_attestation","attest_author":"https://pith.science/pith/73IB5S7GCCAUSSMRV2NEG6A7TR/action/author_attestation","sign_citation":"https://pith.science/pith/73IB5S7GCCAUSSMRV2NEG6A7TR/action/citation_signature","submit_replication":"https://pith.science/pith/73IB5S7GCCAUSSMRV2NEG6A7TR/action/replication_record"}},"created_at":"2026-07-05T09:12:02.655381+00:00","updated_at":"2026-07-05T09:12:02.655381+00:00"}