{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:QUSVK6UFCGZJ673V3TPAUQ7KAI","short_pith_number":"pith:QUSVK6UF","schema_version":"1.0","canonical_sha256":"8525557a8511b29f7f75dcde0a43ea02071777681155e240a4633a1c3e377f71","source":{"kind":"arxiv","id":"2508.18076","version":2},"attestation_state":"computed","paper":{"title":"Neither Valid nor Reliable? Investigating the Use of LLMs as Judges","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Golnoosh Farnadi, Jackie Chi Kit Cheung, Khaoula Chehbouni, Mohammed Haddou","submitted_at":"2025-08-25T14:43:10Z","abstract_excerpt":"Evaluating natural language generation (NLG) systems remains a core challenge of natural language processing (NLP), further complicated by the rise of large language models (LLMs) that aims to be general-purpose. Recently, large language models as judges (LLJs) have emerged as a promising alternative to traditional metrics, but their validity remains underexplored. This position paper argues that the current enthusiasm around LLJs may be premature, as their adoption has outpaced rigorous scrutiny of their reliability and validity as evaluators. Drawing on measurement theory from the social sci"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.18076","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-08-25T14:43:10Z","cross_cats_sorted":[],"title_canon_sha256":"6440b8172c8ecde77618761fb3473b63cebb499adcfc90c771d1d1a22e8d6df6","abstract_canon_sha256":"7bea651b6aa1d553d51952b31ebc143cc87014d3184394253ccfbb76880309d2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:00:29.167826Z","signature_b64":"dTQcH3cXNMwIDzPQ5zLBs1pVfCYoBq4WE6i7xkamfw7CR8rZ+9B2K08qE8KZK2JwH2bLZWLXyqCYsvLrbODVBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8525557a8511b29f7f75dcde0a43ea02071777681155e240a4633a1c3e377f71","last_reissued_at":"2026-07-05T12:00:29.167204Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:00:29.167204Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Neither Valid nor Reliable? Investigating the Use of LLMs as Judges","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Golnoosh Farnadi, Jackie Chi Kit Cheung, Khaoula Chehbouni, Mohammed Haddou","submitted_at":"2025-08-25T14:43:10Z","abstract_excerpt":"Evaluating natural language generation (NLG) systems remains a core challenge of natural language processing (NLP), further complicated by the rise of large language models (LLMs) that aims to be general-purpose. Recently, large language models as judges (LLJs) have emerged as a promising alternative to traditional metrics, but their validity remains underexplored. This position paper argues that the current enthusiasm around LLJs may be premature, as their adoption has outpaced rigorous scrutiny of their reliability and validity as evaluators. Drawing on measurement theory from the social sci"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.18076","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.18076/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.18076","created_at":"2026-07-05T12:00:29.167292+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.18076v2","created_at":"2026-07-05T12:00:29.167292+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.18076","created_at":"2026-07-05T12:00:29.167292+00:00"},{"alias_kind":"pith_short_12","alias_value":"QUSVK6UFCGZJ","created_at":"2026-07-05T12:00:29.167292+00:00"},{"alias_kind":"pith_short_16","alias_value":"QUSVK6UFCGZJ673V","created_at":"2026-07-05T12:00:29.167292+00:00"},{"alias_kind":"pith_short_8","alias_value":"QUSVK6UF","created_at":"2026-07-05T12:00:29.167292+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09118","citing_title":"ComplexConstraints and Beyond: Expert Rubrics for RLVR","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02258","citing_title":"Matter to Mechanism: A Benchmark for AI Co-Scientists in Materials and Battery Research","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02406","citing_title":"Evaluating AI-Generated Images of Cultural Artifacts with Community-Informed Rubrics","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16354","citing_title":"Augmenting Human Evaluation with LLM Judges: How Many Human Reviews Do You Need?","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19141","citing_title":"GRASP: Deterministic argument ranking in interaction graphs","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02406","citing_title":"Evaluating AI-Generated Images of Cultural Artifacts with Community-Informed Rubrics","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21102","citing_title":"Leveraging Multimodal LLMs for Built Environment and Housing Attribute Assessment from Street-View Imagery","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QUSVK6UFCGZJ673V3TPAUQ7KAI","json":"https://pith.science/pith/QUSVK6UFCGZJ673V3TPAUQ7KAI.json","graph_json":"https://pith.science/api/pith-number/QUSVK6UFCGZJ673V3TPAUQ7KAI/graph.json","events_json":"https://pith.science/api/pith-number/QUSVK6UFCGZJ673V3TPAUQ7KAI/events.json","paper":"https://pith.science/paper/QUSVK6UF"},"agent_actions":{"view_html":"https://pith.science/pith/QUSVK6UFCGZJ673V3TPAUQ7KAI","download_json":"https://pith.science/pith/QUSVK6UFCGZJ673V3TPAUQ7KAI.json","view_paper":"https://pith.science/paper/QUSVK6UF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.18076&json=true","fetch_graph":"https://pith.science/api/pith-number/QUSVK6UFCGZJ673V3TPAUQ7KAI/graph.json","fetch_events":"https://pith.science/api/pith-number/QUSVK6UFCGZJ673V3TPAUQ7KAI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QUSVK6UFCGZJ673V3TPAUQ7KAI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QUSVK6UFCGZJ673V3TPAUQ7KAI/action/storage_attestation","attest_author":"https://pith.science/pith/QUSVK6UFCGZJ673V3TPAUQ7KAI/action/author_attestation","sign_citation":"https://pith.science/pith/QUSVK6UFCGZJ673V3TPAUQ7KAI/action/citation_signature","submit_replication":"https://pith.science/pith/QUSVK6UFCGZJ673V3TPAUQ7KAI/action/replication_record"}},"created_at":"2026-07-05T12:00:29.167292+00:00","updated_at":"2026-07-05T12:00:29.167292+00:00"}