{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:NOBPECM6ILD5AFVJ65Z4WDWNLM","short_pith_number":"pith:NOBPECM6","schema_version":"1.0","canonical_sha256":"6b82f2099e42c7d016a9f773cb0ecd5b194be725c7aca016fe6e4e39b0c4eee4","source":{"kind":"arxiv","id":"2412.12509","version":2},"attestation_state":"computed","paper":{"title":"Can You Trust LLM Judgments? Reliability of LLM-as-a-Judge","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Kayla Schroeder, Zach Wood-Doughty","submitted_at":"2024-12-17T03:37:31Z","abstract_excerpt":"Large Language Models (LLMs) have become increasingly powerful and ubiquitous, but their stochastic nature poses challenges to the reliability of their outputs. While deterministic settings can improve consistency, they do not guarantee reliability, as a single sample from the model's probability distribution can still be misleading. Building upon the concept of LLM-as-a-judge, we introduce a novel framework for rigorously evaluating the reliability of LLM judgments, leveraging McDonald's omega. We evaluate the reliability of LLMs when judging the outputs of other LLMs on standard single-turn "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.12509","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-12-17T03:37:31Z","cross_cats_sorted":[],"title_canon_sha256":"f9bc0e74640a2570bbd2158666fa2705af3365caa36fa0142427c467d336ff65","abstract_canon_sha256":"f7f10f09fb39c44b6e6173eb95f8bf5aab29b00ccdf26623890801b1b6b843de"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:16:18.284855Z","signature_b64":"eBezevdFoqcfxKHyUwRyzt6KaPTttU+e4iaXr3yLH4l5r25jR3lfbGKPemxTNnHvRYuEU8/piTZF3U4hQbLVBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6b82f2099e42c7d016a9f773cb0ecd5b194be725c7aca016fe6e4e39b0c4eee4","last_reissued_at":"2026-07-05T10:16:18.284164Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:16:18.284164Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Can You Trust LLM Judgments? Reliability of LLM-as-a-Judge","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Kayla Schroeder, Zach Wood-Doughty","submitted_at":"2024-12-17T03:37:31Z","abstract_excerpt":"Large Language Models (LLMs) have become increasingly powerful and ubiquitous, but their stochastic nature poses challenges to the reliability of their outputs. While deterministic settings can improve consistency, they do not guarantee reliability, as a single sample from the model's probability distribution can still be misleading. Building upon the concept of LLM-as-a-judge, we introduce a novel framework for rigorously evaluating the reliability of LLM judgments, leveraging McDonald's omega. We evaluate the reliability of LLMs when judging the outputs of other LLMs on standard single-turn "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.12509","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.12509/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.12509","created_at":"2026-07-05T10:16:18.284244+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.12509v2","created_at":"2026-07-05T10:16:18.284244+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.12509","created_at":"2026-07-05T10:16:18.284244+00:00"},{"alias_kind":"pith_short_12","alias_value":"NOBPECM6ILD5","created_at":"2026-07-05T10:16:18.284244+00:00"},{"alias_kind":"pith_short_16","alias_value":"NOBPECM6ILD5AFVJ","created_at":"2026-07-05T10:16:18.284244+00:00"},{"alias_kind":"pith_short_8","alias_value":"NOBPECM6","created_at":"2026-07-05T10:16:18.284244+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":17,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24428","citing_title":"Escaping the Self-Confirmation Trap: An Execute-Distill-Verify Paradigm for Agentic Experience Learning","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09118","citing_title":"ComplexConstraints and Beyond: Expert Rubrics for RLVR","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07032","citing_title":"A Systematic Investigation of RL-Jailbreaking in LLMs","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30987","citing_title":"Measuring Judgment Quality in Natural-Language Explanations: Evidence from Forecasting Tournaments","ref_index":124,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24279","citing_title":"ContextEcho: A Benchmark for Persona Drift in Long Agentic-Coding Sessions","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25440","citing_title":"A Multi-Agent LLM Framework for Rating the Quality of Surgical Feedback","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2508.15503","citing_title":"Guidelines for Empirical Studies in Software Engineering involving Large Language Models","ref_index":118,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19035","citing_title":"Trustworthy Agent Network: Trust in Agent Networks Must Be Baked In, Not Bolted On","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2508.15503","citing_title":"Guidelines for Empirical Studies in Software Engineering involving Large Language Models","ref_index":118,"is_internal_anchor":false},{"citing_arxiv_id":"2512.10687","citing_title":"Safe for Whom? Rethinking How We Evaluate the Safety of LLMs for Real Users","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15326","citing_title":"Analyzing the Presentation, Content, and Utilization of References in LLM-powered Conversational AI Systems","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09121","citing_title":"A Communication-Theoretic Framework for LLM Agents: Cost-Aware Adaptive Reliability","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20652","citing_title":"Large Language Models Outperform Humans in Fraud Detection and Resistance to Motivated Investor Pressure","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07032","citing_title":"A Systematic Investigation of RL-Jailbreaking in LLMs","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05371","citing_title":"LLM-as-Judge for Semantic Judging of Powerline Segmentation in UAV Inspection","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05593","citing_title":"Label Effects: Shared Heuristic Reliance in Trust Assessment by Humans and LLM-as-a-Judge","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16706","citing_title":"Evaluating Tool-Using Language Agents: Judge Reliability, Propagation Cascades, and Runtime Mitigation in AgentProp-Bench","ref_index":20,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NOBPECM6ILD5AFVJ65Z4WDWNLM","json":"https://pith.science/pith/NOBPECM6ILD5AFVJ65Z4WDWNLM.json","graph_json":"https://pith.science/api/pith-number/NOBPECM6ILD5AFVJ65Z4WDWNLM/graph.json","events_json":"https://pith.science/api/pith-number/NOBPECM6ILD5AFVJ65Z4WDWNLM/events.json","paper":"https://pith.science/paper/NOBPECM6"},"agent_actions":{"view_html":"https://pith.science/pith/NOBPECM6ILD5AFVJ65Z4WDWNLM","download_json":"https://pith.science/pith/NOBPECM6ILD5AFVJ65Z4WDWNLM.json","view_paper":"https://pith.science/paper/NOBPECM6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.12509&json=true","fetch_graph":"https://pith.science/api/pith-number/NOBPECM6ILD5AFVJ65Z4WDWNLM/graph.json","fetch_events":"https://pith.science/api/pith-number/NOBPECM6ILD5AFVJ65Z4WDWNLM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NOBPECM6ILD5AFVJ65Z4WDWNLM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NOBPECM6ILD5AFVJ65Z4WDWNLM/action/storage_attestation","attest_author":"https://pith.science/pith/NOBPECM6ILD5AFVJ65Z4WDWNLM/action/author_attestation","sign_citation":"https://pith.science/pith/NOBPECM6ILD5AFVJ65Z4WDWNLM/action/citation_signature","submit_replication":"https://pith.science/pith/NOBPECM6ILD5AFVJ65Z4WDWNLM/action/replication_record"}},"created_at":"2026-07-05T10:16:18.284244+00:00","updated_at":"2026-07-05T10:16:18.284244+00:00"}