{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ZE2RC2GNXWDNNO6MA2A6EJ2FYZ","short_pith_number":"pith:ZE2RC2GN","schema_version":"1.0","canonical_sha256":"c9351168cdbd86d6bbcc0681e22745c64db66a65ad32cd653f8041054719a4f3","source":{"kind":"arxiv","id":"2402.14860","version":4},"attestation_state":"computed","paper":{"title":"Ranking Large Language Models without Ground Truth","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Amit Dhurandhar, Elizabeth Daly, Karthikeyan Natesan Ramamurthy, Moninder Singh, Rahul Nair","submitted_at":"2024-02-21T00:49:43Z","abstract_excerpt":"Evaluation and ranking of large language models (LLMs) has become an important problem with the proliferation of these models and their impact. Evaluation methods either require human responses which are expensive to acquire or use pairs of LLMs to evaluate each other which can be unreliable. In this paper, we provide a novel perspective where, given a dataset of prompts (viz. questions, instructions, etc.) and a set of LLMs, we rank them without access to any ground truth or reference responses. Inspired by real life where both an expert and a knowledgeable person can identify a novice our ma"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.14860","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-02-21T00:49:43Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"aacd57dacd37ce27e1ad384a92d8f8a79bc7327044144d38d2c3f5a38eabaf06","abstract_canon_sha256":"482571bf4f07257ccc2d6fb417271a5bbec970cac80ea6d46c95df1f0606d2b6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:29:22.890491Z","signature_b64":"ll7vRQYNAPon0M+8VnWjURvbiBlEUHv9usRpZDEQ/W0KM57lqIlbVI27BOrVHBNpb+FlLIj8iNQE/9lk59qcCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c9351168cdbd86d6bbcc0681e22745c64db66a65ad32cd653f8041054719a4f3","last_reissued_at":"2026-07-05T08:29:22.890000Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:29:22.890000Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Ranking Large Language Models without Ground Truth","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Amit Dhurandhar, Elizabeth Daly, Karthikeyan Natesan Ramamurthy, Moninder Singh, Rahul Nair","submitted_at":"2024-02-21T00:49:43Z","abstract_excerpt":"Evaluation and ranking of large language models (LLMs) has become an important problem with the proliferation of these models and their impact. Evaluation methods either require human responses which are expensive to acquire or use pairs of LLMs to evaluate each other which can be unreliable. In this paper, we provide a novel perspective where, given a dataset of prompts (viz. questions, instructions, etc.) and a set of LLMs, we rank them without access to any ground truth or reference responses. Inspired by real life where both an expert and a knowledgeable person can identify a novice our ma"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.14860","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.14860/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.14860","created_at":"2026-07-05T08:29:22.890060+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.14860v4","created_at":"2026-07-05T08:29:22.890060+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.14860","created_at":"2026-07-05T08:29:22.890060+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZE2RC2GNXWDN","created_at":"2026-07-05T08:29:22.890060+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZE2RC2GNXWDNNO6M","created_at":"2026-07-05T08:29:22.890060+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZE2RC2GN","created_at":"2026-07-05T08:29:22.890060+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2509.01847","citing_title":"Uncertainty Quantification for Ranking with Heterogeneous Preferences","ref_index":27,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZE2RC2GNXWDNNO6MA2A6EJ2FYZ","json":"https://pith.science/pith/ZE2RC2GNXWDNNO6MA2A6EJ2FYZ.json","graph_json":"https://pith.science/api/pith-number/ZE2RC2GNXWDNNO6MA2A6EJ2FYZ/graph.json","events_json":"https://pith.science/api/pith-number/ZE2RC2GNXWDNNO6MA2A6EJ2FYZ/events.json","paper":"https://pith.science/paper/ZE2RC2GN"},"agent_actions":{"view_html":"https://pith.science/pith/ZE2RC2GNXWDNNO6MA2A6EJ2FYZ","download_json":"https://pith.science/pith/ZE2RC2GNXWDNNO6MA2A6EJ2FYZ.json","view_paper":"https://pith.science/paper/ZE2RC2GN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.14860&json=true","fetch_graph":"https://pith.science/api/pith-number/ZE2RC2GNXWDNNO6MA2A6EJ2FYZ/graph.json","fetch_events":"https://pith.science/api/pith-number/ZE2RC2GNXWDNNO6MA2A6EJ2FYZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZE2RC2GNXWDNNO6MA2A6EJ2FYZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZE2RC2GNXWDNNO6MA2A6EJ2FYZ/action/storage_attestation","attest_author":"https://pith.science/pith/ZE2RC2GNXWDNNO6MA2A6EJ2FYZ/action/author_attestation","sign_citation":"https://pith.science/pith/ZE2RC2GNXWDNNO6MA2A6EJ2FYZ/action/citation_signature","submit_replication":"https://pith.science/pith/ZE2RC2GNXWDNNO6MA2A6EJ2FYZ/action/replication_record"}},"created_at":"2026-07-05T08:29:22.890060+00:00","updated_at":"2026-07-05T08:29:22.890060+00:00"}