{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:2BN54A6D2I2BONFPIDYGJBKUFI","short_pith_number":"pith:2BN54A6D","schema_version":"1.0","canonical_sha256":"d05bde03c3d2341734af40f06485542a16c621ec1b78b79742d4c102d5e3c501","source":{"kind":"arxiv","id":"2307.03025","version":3},"attestation_state":"computed","paper":{"title":"Style Over Substance: Evaluation Biases for Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alham Fikri Aji, Minghao Wu","submitted_at":"2023-07-06T14:42:01Z","abstract_excerpt":"As large language models (LLMs) continue to advance, accurately and comprehensively evaluating their performance becomes increasingly challenging. Ranking the relative performance of LLMs based on Elo ratings, according to human judgment, is gaining more popularity. However, the extent to which humans and LLMs are capable evaluators remains uncertain. This study investigates the behavior of crowd-sourced and expert annotators, as well as LLMs, when comparing outputs from different models. To achieve this, we curate a dataset of intentionally flawed machine-generated answers. Our findings revea"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.03025","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-07-06T14:42:01Z","cross_cats_sorted":[],"title_canon_sha256":"969313b84bf49bebc4bb3e3304354340037da0c6620e2db2fee3bccc33e7fbc6","abstract_canon_sha256":"3305c8676fe0a73083ee1227aea2aae857d688f7fb004d8aadaaa92b5c0dab47"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:11:57.334880Z","signature_b64":"XVag9QIRR4yi6QukY/d0aBHac5Y/ouuVTz+scqkfAdVzc+rK1Y+X1WlxCZF1xncCOXNRfNKXQr9jeipVU9N7Cw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d05bde03c3d2341734af40f06485542a16c621ec1b78b79742d4c102d5e3c501","last_reissued_at":"2026-07-05T07:11:57.334387Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:11:57.334387Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Style Over Substance: Evaluation Biases for Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alham Fikri Aji, Minghao Wu","submitted_at":"2023-07-06T14:42:01Z","abstract_excerpt":"As large language models (LLMs) continue to advance, accurately and comprehensively evaluating their performance becomes increasingly challenging. Ranking the relative performance of LLMs based on Elo ratings, according to human judgment, is gaining more popularity. However, the extent to which humans and LLMs are capable evaluators remains uncertain. This study investigates the behavior of crowd-sourced and expert annotators, as well as LLMs, when comparing outputs from different models. To achieve this, we curate a dataset of intentionally flawed machine-generated answers. Our findings revea"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.03025","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.03025/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.03025","created_at":"2026-07-05T07:11:57.334456+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.03025v3","created_at":"2026-07-05T07:11:57.334456+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.03025","created_at":"2026-07-05T07:11:57.334456+00:00"},{"alias_kind":"pith_short_12","alias_value":"2BN54A6D2I2B","created_at":"2026-07-05T07:11:57.334456+00:00"},{"alias_kind":"pith_short_16","alias_value":"2BN54A6D2I2BONFP","created_at":"2026-07-05T07:11:57.334456+00:00"},{"alias_kind":"pith_short_8","alias_value":"2BN54A6D","created_at":"2026-07-05T07:11:57.334456+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2405.14782","citing_title":"Lessons from the Trenches on Reproducible Evaluation of Language Models","ref_index":87,"is_internal_anchor":false},{"citing_arxiv_id":"2410.02736","citing_title":"Justice or Prejudice? Quantifying Biases in LLM-as-a-Judge","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23178","citing_title":"Judging the Judges: A Systematic Evaluation of Bias Mitigation Strategies in LLM-as-a-Judge Pipelines","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01123","citing_title":"PERSA: Reinforcement Learning for Professor-Style Personalized Feedback with LLMs","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07709","citing_title":"IatroBench: Pre-Registered Evidence of Iatrogenic Harm from AI Safety Measures","ref_index":36,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2BN54A6D2I2BONFPIDYGJBKUFI","json":"https://pith.science/pith/2BN54A6D2I2BONFPIDYGJBKUFI.json","graph_json":"https://pith.science/api/pith-number/2BN54A6D2I2BONFPIDYGJBKUFI/graph.json","events_json":"https://pith.science/api/pith-number/2BN54A6D2I2BONFPIDYGJBKUFI/events.json","paper":"https://pith.science/paper/2BN54A6D"},"agent_actions":{"view_html":"https://pith.science/pith/2BN54A6D2I2BONFPIDYGJBKUFI","download_json":"https://pith.science/pith/2BN54A6D2I2BONFPIDYGJBKUFI.json","view_paper":"https://pith.science/paper/2BN54A6D","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.03025&json=true","fetch_graph":"https://pith.science/api/pith-number/2BN54A6D2I2BONFPIDYGJBKUFI/graph.json","fetch_events":"https://pith.science/api/pith-number/2BN54A6D2I2BONFPIDYGJBKUFI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2BN54A6D2I2BONFPIDYGJBKUFI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2BN54A6D2I2BONFPIDYGJBKUFI/action/storage_attestation","attest_author":"https://pith.science/pith/2BN54A6D2I2BONFPIDYGJBKUFI/action/author_attestation","sign_citation":"https://pith.science/pith/2BN54A6D2I2BONFPIDYGJBKUFI/action/citation_signature","submit_replication":"https://pith.science/pith/2BN54A6D2I2BONFPIDYGJBKUFI/action/replication_record"}},"created_at":"2026-07-05T07:11:57.334456+00:00","updated_at":"2026-07-05T07:11:57.334456+00:00"}