{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:P67NIWZLVJIBPLLP75QX43MXWR","short_pith_number":"pith:P67NIWZL","schema_version":"1.0","canonical_sha256":"7fbed45b2baa5017ad6fff617e6d97b44286d6e958b6718dd30592da08cce352","source":{"kind":"arxiv","id":"2607.21632","version":1},"attestation_state":"computed","paper":{"title":"A Consensus-Based Framework for Relative Preference Evaluation of Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Mohtashim Khan","submitted_at":"2026-07-19T08:59:19Z","abstract_excerpt":"Traditional benchmarks for LLMs primarily rely on static datasets and objective scoring metrics, which often fail to capture differences in response quality when multiple answers are acceptable. In such settings, correctness alone is insufficient to distinguish between responses that vary in clarity, completeness, and usefulness.\n  This paper introduces a consensus-based evaluation framework that measures relative preference among model-generated responses rather than absolute correctness. Instead of evaluating outputs against a fixed ground truth, we assess how a panel of diverse LLMs ranks a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.21632","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2026-07-19T08:59:19Z","cross_cats_sorted":[],"title_canon_sha256":"5fcd174c91761b268d9475999b91198a29cbbdc4b99184116f3312f55d86bf7e","abstract_canon_sha256":"a5699fccf5c8dc07971c379b93efe950a91341df84ebee11fe057ecf55bfb18d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-27T00:20:12.030136Z","signature_b64":"B2zSHZgqHQpo2Bxwn0nJwhCE9CTGrigSKtIbz/5TI7/f6xzWtlxs0WUtpJVm4r9vfqIJVk18jjUBUWZiOcsSAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7fbed45b2baa5017ad6fff617e6d97b44286d6e958b6718dd30592da08cce352","last_reissued_at":"2026-07-27T00:20:12.029311Z","signature_status":"signed_v1","first_computed_at":"2026-07-27T00:20:12.029311Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Consensus-Based Framework for Relative Preference Evaluation of Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Mohtashim Khan","submitted_at":"2026-07-19T08:59:19Z","abstract_excerpt":"Traditional benchmarks for LLMs primarily rely on static datasets and objective scoring metrics, which often fail to capture differences in response quality when multiple answers are acceptable. In such settings, correctness alone is insufficient to distinguish between responses that vary in clarity, completeness, and usefulness.\n  This paper introduces a consensus-based evaluation framework that measures relative preference among model-generated responses rather than absolute correctness. Instead of evaluating outputs against a fixed ground truth, we assess how a panel of diverse LLMs ranks a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.21632","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.21632/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.21632","created_at":"2026-07-27T00:20:12.029752+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.21632v1","created_at":"2026-07-27T00:20:12.029752+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.21632","created_at":"2026-07-27T00:20:12.029752+00:00"},{"alias_kind":"pith_short_12","alias_value":"P67NIWZLVJIB","created_at":"2026-07-27T00:20:12.029752+00:00"},{"alias_kind":"pith_short_16","alias_value":"P67NIWZLVJIBPLLP","created_at":"2026-07-27T00:20:12.029752+00:00"},{"alias_kind":"pith_short_8","alias_value":"P67NIWZL","created_at":"2026-07-27T00:20:12.029752+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/P67NIWZLVJIBPLLP75QX43MXWR","json":"https://pith.science/pith/P67NIWZLVJIBPLLP75QX43MXWR.json","graph_json":"https://pith.science/api/pith-number/P67NIWZLVJIBPLLP75QX43MXWR/graph.json","events_json":"https://pith.science/api/pith-number/P67NIWZLVJIBPLLP75QX43MXWR/events.json","paper":"https://pith.science/paper/P67NIWZL"},"agent_actions":{"view_html":"https://pith.science/pith/P67NIWZLVJIBPLLP75QX43MXWR","download_json":"https://pith.science/pith/P67NIWZLVJIBPLLP75QX43MXWR.json","view_paper":"https://pith.science/paper/P67NIWZL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.21632&json=true","fetch_graph":"https://pith.science/api/pith-number/P67NIWZLVJIBPLLP75QX43MXWR/graph.json","fetch_events":"https://pith.science/api/pith-number/P67NIWZLVJIBPLLP75QX43MXWR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/P67NIWZLVJIBPLLP75QX43MXWR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/P67NIWZLVJIBPLLP75QX43MXWR/action/storage_attestation","attest_author":"https://pith.science/pith/P67NIWZLVJIBPLLP75QX43MXWR/action/author_attestation","sign_citation":"https://pith.science/pith/P67NIWZLVJIBPLLP75QX43MXWR/action/citation_signature","submit_replication":"https://pith.science/pith/P67NIWZLVJIBPLLP75QX43MXWR/action/replication_record"}},"created_at":"2026-07-27T00:20:12.029752+00:00","updated_at":"2026-07-27T00:20:12.029752+00:00"}