{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:UZBW7OHEZJPNORQIYDBEZDOYLE","short_pith_number":"pith:UZBW7OHE","schema_version":"1.0","canonical_sha256":"a6436fb8e4ca5ed74608c0c24c8dd8593994067e70d357b7e5f4159ac1148056","source":{"kind":"arxiv","id":"2310.10076","version":1},"attestation_state":"computed","paper":{"title":"Verbosity Bias in Preference Labeling by Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Akifumi Wachi, Keita Saito, Koki Wataoka, Youhei Akimoto","submitted_at":"2023-10-16T05:19:02Z","abstract_excerpt":"In recent years, Large Language Models (LLMs) have witnessed a remarkable surge in prevalence, altering the landscape of natural language processing and machine learning. One key factor in improving the performance of LLMs is alignment with humans achieved with Reinforcement Learning from Human Feedback (RLHF), as for many LLMs such as GPT-4, Bard, etc. In addition, recent studies are investigating the replacement of human feedback with feedback from other LLMs named Reinforcement Learning from AI Feedback (RLAIF). We examine the biases that come along with evaluating LLMs with other LLMs and "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.10076","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-10-16T05:19:02Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"df0073455e32a0442059d6d0f873ac6efcd95f36c58d34559d1d02db0d189f5a","abstract_canon_sha256":"fc13d756826b40f7b600a9e2a0d673a588ec4c2bac2eb3f99da0b0515b530ece"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:01:14.886832Z","signature_b64":"0Xw4uZxlfWEwouApqrKTBCie3Vgjht9muec+3R/+23JQ6VZxuKjjhLdXAMAhey8YQqpi9A5Hzg+22PlokBS7Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a6436fb8e4ca5ed74608c0c24c8dd8593994067e70d357b7e5f4159ac1148056","last_reissued_at":"2026-07-05T07:01:14.886357Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:01:14.886357Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Verbosity Bias in Preference Labeling by Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Akifumi Wachi, Keita Saito, Koki Wataoka, Youhei Akimoto","submitted_at":"2023-10-16T05:19:02Z","abstract_excerpt":"In recent years, Large Language Models (LLMs) have witnessed a remarkable surge in prevalence, altering the landscape of natural language processing and machine learning. One key factor in improving the performance of LLMs is alignment with humans achieved with Reinforcement Learning from Human Feedback (RLHF), as for many LLMs such as GPT-4, Bard, etc. In addition, recent studies are investigating the replacement of human feedback with feedback from other LLMs named Reinforcement Learning from AI Feedback (RLAIF). We examine the biases that come along with evaluating LLMs with other LLMs and "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.10076","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.10076/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.10076","created_at":"2026-07-05T07:01:14.886414+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.10076v1","created_at":"2026-07-05T07:01:14.886414+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.10076","created_at":"2026-07-05T07:01:14.886414+00:00"},{"alias_kind":"pith_short_12","alias_value":"UZBW7OHEZJPN","created_at":"2026-07-05T07:01:14.886414+00:00"},{"alias_kind":"pith_short_16","alias_value":"UZBW7OHEZJPNORQI","created_at":"2026-07-05T07:01:14.886414+00:00"},{"alias_kind":"pith_short_8","alias_value":"UZBW7OHE","created_at":"2026-07-05T07:01:14.886414+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":19,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08535","citing_title":"When the Judge Changes, So Does the Measurement: Auditing LLM-as-Judge Reliability","ref_index":18,"is_internal_anchor":true},{"citing_arxiv_id":"2606.09843","citing_title":"An LLM-Native Psychometric Instrument Reveals a Self-Report--Behavior Gap Across 25 Models","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22329","citing_title":"BabelJudge: Measuring LLM-as-a-Judge Reliability Across Languages and Agent Trajectories","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19057","citing_title":"Quantifying and Auditing LLM Evaluation via Positive--Unlabeled Learning","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02104","citing_title":"When Can You Debias an LLM Judge? Identifiability Limits, a Test, and Designs for Top-k Ranking","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05384","citing_title":"Stability vs. Manipulability: Evaluating Robustness Under Post-Decision Interaction in LLM Judges","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30412","citing_title":"Can LLMs Rank? A Tale of Triads and Triage","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30931","citing_title":"RoPoLL: Robust Panel of LLM Judges","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10379","citing_title":"Not All Proofs Are Equal: Evaluating LLM Proof Quality Beyond Correctness","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24699","citing_title":"MDIA: A Multi-Agent Diagnostic Intelligence Pipeline on HealthBench Professional","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30529","citing_title":"Generalistic or Specific Embeddings, Which is Better? An Empirical Study on Search for Clinical Coding in Non-English Languages","ref_index":93,"is_internal_anchor":false},{"citing_arxiv_id":"2411.15594","citing_title":"A Survey on LLM-as-a-Judge","ref_index":124,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19099","citing_title":"DecisionBench: A Benchmark for Emergent Delegation in Long-Horizon Agentic Workflows","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2602.17170","citing_title":"When LLM Judges Inflate Scores: Exploring Overrating in Relevance Assessment","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11134","citing_title":"Spurious Correlation Learning in Preference Optimization: Mechanisms, Consequences, and Mitigation via Tie Training","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10379","citing_title":"Not All Proofs Are Equal: Evaluating LLM Proof Quality Beyond Correctness","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23178","citing_title":"Judging the Judges: A Systematic Evaluation of Bias Mitigation Strategies in LLM-as-a-Judge Pipelines","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26965","citing_title":"The Impact of AI-Generated Text on the Internet","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15302","citing_title":"Diagnosing LLM Judge Reliability: Conformal Prediction Sets and Transitivity Violations","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UZBW7OHEZJPNORQIYDBEZDOYLE","json":"https://pith.science/pith/UZBW7OHEZJPNORQIYDBEZDOYLE.json","graph_json":"https://pith.science/api/pith-number/UZBW7OHEZJPNORQIYDBEZDOYLE/graph.json","events_json":"https://pith.science/api/pith-number/UZBW7OHEZJPNORQIYDBEZDOYLE/events.json","paper":"https://pith.science/paper/UZBW7OHE"},"agent_actions":{"view_html":"https://pith.science/pith/UZBW7OHEZJPNORQIYDBEZDOYLE","download_json":"https://pith.science/pith/UZBW7OHEZJPNORQIYDBEZDOYLE.json","view_paper":"https://pith.science/paper/UZBW7OHE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.10076&json=true","fetch_graph":"https://pith.science/api/pith-number/UZBW7OHEZJPNORQIYDBEZDOYLE/graph.json","fetch_events":"https://pith.science/api/pith-number/UZBW7OHEZJPNORQIYDBEZDOYLE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UZBW7OHEZJPNORQIYDBEZDOYLE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UZBW7OHEZJPNORQIYDBEZDOYLE/action/storage_attestation","attest_author":"https://pith.science/pith/UZBW7OHEZJPNORQIYDBEZDOYLE/action/author_attestation","sign_citation":"https://pith.science/pith/UZBW7OHEZJPNORQIYDBEZDOYLE/action/citation_signature","submit_replication":"https://pith.science/pith/UZBW7OHEZJPNORQIYDBEZDOYLE/action/replication_record"}},"created_at":"2026-07-05T07:01:14.886414+00:00","updated_at":"2026-07-05T07:01:14.886414+00:00"}