{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:KCKIZSDPY6W6FPGIN5VCB6G5XF","short_pith_number":"pith:KCKIZSDP","schema_version":"1.0","canonical_sha256":"50948cc86fc7ade2bcc86f6a20f8ddb96afa2c471f87daac37497c7ca9bc4b03","source":{"kind":"arxiv","id":"2506.02592","version":1},"attestation_state":"computed","paper":{"title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Enrui Hu, Hao Wang, Xinyu Zhang, Yankai Lin, Zhi-Yuan Chen","submitted_at":"2025-06-03T08:12:47Z","abstract_excerpt":"Recent studies show that large language models (LLMs) exhibit self-preference bias when serving as judges, meaning they tend to favor their own responses over those generated by other models. Existing methods typically measure this bias by calculating the difference between the scores a judge model assigns to its own responses and those it assigns to responses from other models. However, this approach conflates self-preference bias with response quality, as higher-quality responses from the judge model may also lead to positive score differences, even in the absence of bias. To address this is"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.02592","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-06-03T08:12:47Z","cross_cats_sorted":[],"title_canon_sha256":"85ff6a561ff42e6240b51a9c96763f3078d8fcf826d71d2b6c7581fec3306f69","abstract_canon_sha256":"16b53686bbd2ee981eeb395820e0b1ab48e71e0ef276aefe2fc86fd2eba12e43"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:15:04.955356Z","signature_b64":"+xL3Mf91UiavMI22c5TgC3kTSrXZW0f2q1KAMpGFxrvZ+pG9YUQacbCp3eNsaP8rs+uXwMERVrvwDRNHXuYVBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"50948cc86fc7ade2bcc86f6a20f8ddb96afa2c471f87daac37497c7ca9bc4b03","last_reissued_at":"2026-07-05T11:15:04.954844Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:15:04.954844Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Beyond the Surface: Measuring Self-Preference in LLM Judgments","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Enrui Hu, Hao Wang, Xinyu Zhang, Yankai Lin, Zhi-Yuan Chen","submitted_at":"2025-06-03T08:12:47Z","abstract_excerpt":"Recent studies show that large language models (LLMs) exhibit self-preference bias when serving as judges, meaning they tend to favor their own responses over those generated by other models. Existing methods typically measure this bias by calculating the difference between the scores a judge model assigns to its own responses and those it assigns to responses from other models. However, this approach conflates self-preference bias with response quality, as higher-quality responses from the judge model may also lead to positive score differences, even in the absence of bias. To address this is"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.02592","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.02592/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.02592","created_at":"2026-07-05T11:15:04.954903+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.02592v1","created_at":"2026-07-05T11:15:04.954903+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.02592","created_at":"2026-07-05T11:15:04.954903+00:00"},{"alias_kind":"pith_short_12","alias_value":"KCKIZSDPY6W6","created_at":"2026-07-05T11:15:04.954903+00:00"},{"alias_kind":"pith_short_16","alias_value":"KCKIZSDPY6W6FPGI","created_at":"2026-07-05T11:15:04.954903+00:00"},{"alias_kind":"pith_short_8","alias_value":"KCKIZSDP","created_at":"2026-07-05T11:15:04.954903+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2509.26464","citing_title":"Extreme Self-Preference in Language Models","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2510.07517","citing_title":"When Identity Skews Debate: Anonymization for Bias-Reduced Multi-Agent Reasoning","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KCKIZSDPY6W6FPGIN5VCB6G5XF","json":"https://pith.science/pith/KCKIZSDPY6W6FPGIN5VCB6G5XF.json","graph_json":"https://pith.science/api/pith-number/KCKIZSDPY6W6FPGIN5VCB6G5XF/graph.json","events_json":"https://pith.science/api/pith-number/KCKIZSDPY6W6FPGIN5VCB6G5XF/events.json","paper":"https://pith.science/paper/KCKIZSDP"},"agent_actions":{"view_html":"https://pith.science/pith/KCKIZSDPY6W6FPGIN5VCB6G5XF","download_json":"https://pith.science/pith/KCKIZSDPY6W6FPGIN5VCB6G5XF.json","view_paper":"https://pith.science/paper/KCKIZSDP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.02592&json=true","fetch_graph":"https://pith.science/api/pith-number/KCKIZSDPY6W6FPGIN5VCB6G5XF/graph.json","fetch_events":"https://pith.science/api/pith-number/KCKIZSDPY6W6FPGIN5VCB6G5XF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KCKIZSDPY6W6FPGIN5VCB6G5XF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KCKIZSDPY6W6FPGIN5VCB6G5XF/action/storage_attestation","attest_author":"https://pith.science/pith/KCKIZSDPY6W6FPGIN5VCB6G5XF/action/author_attestation","sign_citation":"https://pith.science/pith/KCKIZSDPY6W6FPGIN5VCB6G5XF/action/citation_signature","submit_replication":"https://pith.science/pith/KCKIZSDPY6W6FPGIN5VCB6G5XF/action/replication_record"}},"created_at":"2026-07-05T11:15:04.954903+00:00","updated_at":"2026-07-05T11:15:04.954903+00:00"}