{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:62RPXGAZTREPGGU3Q7SEEB732A","short_pith_number":"pith:62RPXGAZ","schema_version":"1.0","canonical_sha256":"f6a2fb98199c48f31a9b87e44207fbd0382877e2fbd94b6bfe8f4687ac47a2fa","source":{"kind":"arxiv","id":"2402.17826","version":3},"attestation_state":"computed","paper":{"title":"Prediction-Powered Ranking of Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CY","cs.HC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Eleni Straitouri, Ivi Chatzi, Manuel Gomez Rodriguez, Suhas Thejaswi","submitted_at":"2024-02-27T19:00:01Z","abstract_excerpt":"Large language models are often ranked according to their level of alignment with human preferences -- a model is better than other models if its outputs are more frequently preferred by humans. One of the popular ways to elicit human preferences utilizes pairwise comparisons between the outputs provided by different models to the same inputs. However, since gathering pairwise comparisons by humans is costly and time-consuming, it has become a common practice to gather pairwise comparisons by a strong large language model -- a model strongly aligned with human preferences. Surprisingly, practi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.17826","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-02-27T19:00:01Z","cross_cats_sorted":["cs.AI","cs.CL","cs.CY","cs.HC","stat.ML"],"title_canon_sha256":"ecc7652ccfe360b09037a84043740ab05b5810fc1bba74f3e81a8040f5595a48","abstract_canon_sha256":"59ac99d4a2bdafddeb10ea0e6afe54adc8b140bf3938e980f744b8ccf08b5a61"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:44:07.797512Z","signature_b64":"/hWB9YnhehNI8WJ29Pa0OvTSBJTjjNGQOD/5wFZjYeEY7BB1QGhBUZ730s3UGQ/4I+0uTo1l6yZrkThHImNOBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f6a2fb98199c48f31a9b87e44207fbd0382877e2fbd94b6bfe8f4687ac47a2fa","last_reissued_at":"2026-07-05T09:44:07.797056Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:44:07.797056Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Prediction-Powered Ranking of Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CY","cs.HC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Eleni Straitouri, Ivi Chatzi, Manuel Gomez Rodriguez, Suhas Thejaswi","submitted_at":"2024-02-27T19:00:01Z","abstract_excerpt":"Large language models are often ranked according to their level of alignment with human preferences -- a model is better than other models if its outputs are more frequently preferred by humans. One of the popular ways to elicit human preferences utilizes pairwise comparisons between the outputs provided by different models to the same inputs. However, since gathering pairwise comparisons by humans is costly and time-consuming, it has become a common practice to gather pairwise comparisons by a strong large language model -- a model strongly aligned with human preferences. Surprisingly, practi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.17826","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.17826/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.17826","created_at":"2026-07-05T09:44:07.797112+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.17826v3","created_at":"2026-07-05T09:44:07.797112+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.17826","created_at":"2026-07-05T09:44:07.797112+00:00"},{"alias_kind":"pith_short_12","alias_value":"62RPXGAZTREP","created_at":"2026-07-05T09:44:07.797112+00:00"},{"alias_kind":"pith_short_16","alias_value":"62RPXGAZTREPGGU3","created_at":"2026-07-05T09:44:07.797112+00:00"},{"alias_kind":"pith_short_8","alias_value":"62RPXGAZ","created_at":"2026-07-05T09:44:07.797112+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.04294","citing_title":"Prediction-Powered E-Values","ref_index":2024,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/62RPXGAZTREPGGU3Q7SEEB732A","json":"https://pith.science/pith/62RPXGAZTREPGGU3Q7SEEB732A.json","graph_json":"https://pith.science/api/pith-number/62RPXGAZTREPGGU3Q7SEEB732A/graph.json","events_json":"https://pith.science/api/pith-number/62RPXGAZTREPGGU3Q7SEEB732A/events.json","paper":"https://pith.science/paper/62RPXGAZ"},"agent_actions":{"view_html":"https://pith.science/pith/62RPXGAZTREPGGU3Q7SEEB732A","download_json":"https://pith.science/pith/62RPXGAZTREPGGU3Q7SEEB732A.json","view_paper":"https://pith.science/paper/62RPXGAZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.17826&json=true","fetch_graph":"https://pith.science/api/pith-number/62RPXGAZTREPGGU3Q7SEEB732A/graph.json","fetch_events":"https://pith.science/api/pith-number/62RPXGAZTREPGGU3Q7SEEB732A/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/62RPXGAZTREPGGU3Q7SEEB732A/action/timestamp_anchor","attest_storage":"https://pith.science/pith/62RPXGAZTREPGGU3Q7SEEB732A/action/storage_attestation","attest_author":"https://pith.science/pith/62RPXGAZTREPGGU3Q7SEEB732A/action/author_attestation","sign_citation":"https://pith.science/pith/62RPXGAZTREPGGU3Q7SEEB732A/action/citation_signature","submit_replication":"https://pith.science/pith/62RPXGAZTREPGGU3Q7SEEB732A/action/replication_record"}},"created_at":"2026-07-05T09:44:07.797112+00:00","updated_at":"2026-07-05T09:44:07.797112+00:00"}