{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:FXHPBFUIJTKRW4QVVMAYRNUU7F","short_pith_number":"pith:FXHPBFUI","schema_version":"1.0","canonical_sha256":"2dcef096884cd51b7215ab0188b694f96869d048c624addfb2fd66d01e76d8fb","source":{"kind":"arxiv","id":"2506.13639","version":1},"attestation_state":"computed","paper":{"title":"An Empirical Study of LLM-as-a-Judge: How Design Choices Impact Evaluation Reliability","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Masafumi Oyamada, Taro Yano, Yusuke Yamauchi","submitted_at":"2025-06-16T16:04:43Z","abstract_excerpt":"As large language models (LLMs) continue to advance, reliable evaluation methods are essential particularly for open-ended, instruction-following tasks. LLM-as-a-Judge enables automatic evaluation using LLMs as evaluators, but its reliability remains uncertain. In this work, we analyze key factors affecting its trustworthiness, focusing on alignment with human judgments and evaluation consistency. Using BIGGENBench and EvalBiasBench, we study the effects of evaluation design, decoding strategies, and Chain-of-Tought (CoT) reasoning in evaluation. Our results show that evaluation criteria are c"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.13639","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-16T16:04:43Z","cross_cats_sorted":[],"title_canon_sha256":"614c53087853a07e05d8935d0fe3da69f56b429a11df82c90f1ab4a0318314d6","abstract_canon_sha256":"bb7abc1df6487aa43c99d68f1d4572a3789a2ee3a902c73c6f10375e71b3cdd8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:22:23.822607Z","signature_b64":"0Z9c/fcjQMUsIqFjBZJWwKeOTQ2wg4MXgzTKgWpY+TRF83s+3DbgOHE1UGtt2OeXWp0TmqE5ulDEW92mFGJICA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2dcef096884cd51b7215ab0188b694f96869d048c624addfb2fd66d01e76d8fb","last_reissued_at":"2026-07-05T11:22:23.822047Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:22:23.822047Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"An Empirical Study of LLM-as-a-Judge: How Design Choices Impact Evaluation Reliability","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Masafumi Oyamada, Taro Yano, Yusuke Yamauchi","submitted_at":"2025-06-16T16:04:43Z","abstract_excerpt":"As large language models (LLMs) continue to advance, reliable evaluation methods are essential particularly for open-ended, instruction-following tasks. LLM-as-a-Judge enables automatic evaluation using LLMs as evaluators, but its reliability remains uncertain. In this work, we analyze key factors affecting its trustworthiness, focusing on alignment with human judgments and evaluation consistency. Using BIGGENBench and EvalBiasBench, we study the effects of evaluation design, decoding strategies, and Chain-of-Tought (CoT) reasoning in evaluation. Our results show that evaluation criteria are c"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.13639","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.13639/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.13639","created_at":"2026-07-05T11:22:23.822118+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.13639v1","created_at":"2026-07-05T11:22:23.822118+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.13639","created_at":"2026-07-05T11:22:23.822118+00:00"},{"alias_kind":"pith_short_12","alias_value":"FXHPBFUIJTKR","created_at":"2026-07-05T11:22:23.822118+00:00"},{"alias_kind":"pith_short_16","alias_value":"FXHPBFUIJTKRW4QV","created_at":"2026-07-05T11:22:23.822118+00:00"},{"alias_kind":"pith_short_8","alias_value":"FXHPBFUI","created_at":"2026-07-05T11:22:23.822118+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08535","citing_title":"When the Judge Changes, So Does the Measurement: Auditing LLM-as-Judge Reliability","ref_index":22,"is_internal_anchor":true},{"citing_arxiv_id":"2606.11635","citing_title":"Are LLMs Bad at Moral Reasoning?","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09118","citing_title":"ComplexConstraints and Beyond: Expert Rubrics for RLVR","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08867","citing_title":"Building Customer Support AI Agents at 100M-User Scale: An Evaluation-Driven Framework","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24319","citing_title":"Omissive Bias in Religious Representation: Benchmarking LLM Answers to Everyday Ethical Decision-making","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15455","citing_title":"Multi-Turn Neural Transparency: Surfacing Neural Activations Improves User Calibration to LLM Behavioral Drift","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09121","citing_title":"A Communication-Theoretic Framework for LLM Agents: Cost-Aware Adaptive Reliability","ref_index":77,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FXHPBFUIJTKRW4QVVMAYRNUU7F","json":"https://pith.science/pith/FXHPBFUIJTKRW4QVVMAYRNUU7F.json","graph_json":"https://pith.science/api/pith-number/FXHPBFUIJTKRW4QVVMAYRNUU7F/graph.json","events_json":"https://pith.science/api/pith-number/FXHPBFUIJTKRW4QVVMAYRNUU7F/events.json","paper":"https://pith.science/paper/FXHPBFUI"},"agent_actions":{"view_html":"https://pith.science/pith/FXHPBFUIJTKRW4QVVMAYRNUU7F","download_json":"https://pith.science/pith/FXHPBFUIJTKRW4QVVMAYRNUU7F.json","view_paper":"https://pith.science/paper/FXHPBFUI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.13639&json=true","fetch_graph":"https://pith.science/api/pith-number/FXHPBFUIJTKRW4QVVMAYRNUU7F/graph.json","fetch_events":"https://pith.science/api/pith-number/FXHPBFUIJTKRW4QVVMAYRNUU7F/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FXHPBFUIJTKRW4QVVMAYRNUU7F/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FXHPBFUIJTKRW4QVVMAYRNUU7F/action/storage_attestation","attest_author":"https://pith.science/pith/FXHPBFUIJTKRW4QVVMAYRNUU7F/action/author_attestation","sign_citation":"https://pith.science/pith/FXHPBFUIJTKRW4QVVMAYRNUU7F/action/citation_signature","submit_replication":"https://pith.science/pith/FXHPBFUIJTKRW4QVVMAYRNUU7F/action/replication_record"}},"created_at":"2026-07-05T11:22:23.822118+00:00","updated_at":"2026-07-05T11:22:23.822118+00:00"}