{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XQBEB6CG44IJMDMTFLPYJVTN5Y","short_pith_number":"pith:XQBEB6CG","schema_version":"1.0","canonical_sha256":"bc0240f846e710960d932adf84d66dee27343b92cffa54383a9263f8e3b44fbf","source":{"kind":"arxiv","id":"2401.15641","version":2},"attestation_state":"computed","paper":{"title":"PRE: A Peer Review Based Large Language Model Evaluator","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.IR","authors_text":"Haitao Li, Qingyao Ai, Yiqun Liu, Yiteng Tu, Zhumin Chu","submitted_at":"2024-01-28T12:33:14Z","abstract_excerpt":"The impressive performance of large language models (LLMs) has attracted considerable attention from the academic and industrial communities. Besides how to construct and train LLMs, how to effectively evaluate and compare the capacity of LLMs has also been well recognized as an important yet difficult problem. Existing paradigms rely on either human annotators or model-based evaluators to evaluate the performance of LLMs on different tasks. However, these paradigms often suffer from high cost, low generalizability, and inherited biases in practice, which make them incapable of supporting the "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.15641","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.IR","submitted_at":"2024-01-28T12:33:14Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"b9f0de85b0b23934bd5e0bb8209afef6984ab3202b34d898c8454f74640d7511","abstract_canon_sha256":"0a24afc828a8245f359f7f522fe1ebaa11feb3e2a81bd32a726ab24303dd47ca"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:26:18.209359Z","signature_b64":"9nKy3MR9+UKoA8mc0vq9EBhbLAgxrcOqmUuXVTPpFg0GRQXs+67nZ3zNSipjL4SvGj3vpLIKKekV9DZvcQHsDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bc0240f846e710960d932adf84d66dee27343b92cffa54383a9263f8e3b44fbf","last_reissued_at":"2026-07-05T08:26:18.208768Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:26:18.208768Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"PRE: A Peer Review Based Large Language Model Evaluator","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.IR","authors_text":"Haitao Li, Qingyao Ai, Yiqun Liu, Yiteng Tu, Zhumin Chu","submitted_at":"2024-01-28T12:33:14Z","abstract_excerpt":"The impressive performance of large language models (LLMs) has attracted considerable attention from the academic and industrial communities. Besides how to construct and train LLMs, how to effectively evaluate and compare the capacity of LLMs has also been well recognized as an important yet difficult problem. Existing paradigms rely on either human annotators or model-based evaluators to evaluate the performance of LLMs on different tasks. However, these paradigms often suffer from high cost, low generalizability, and inherited biases in practice, which make them incapable of supporting the "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.15641","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.15641/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.15641","created_at":"2026-07-05T08:26:18.208833+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.15641v2","created_at":"2026-07-05T08:26:18.208833+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.15641","created_at":"2026-07-05T08:26:18.208833+00:00"},{"alias_kind":"pith_short_12","alias_value":"XQBEB6CG44IJ","created_at":"2026-07-05T08:26:18.208833+00:00"},{"alias_kind":"pith_short_16","alias_value":"XQBEB6CG44IJMDMT","created_at":"2026-07-05T08:26:18.208833+00:00"},{"alias_kind":"pith_short_8","alias_value":"XQBEB6CG","created_at":"2026-07-05T08:26:18.208833+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.30556","citing_title":"Poller: Are LLMs Suitable for Evaluating the Poetry Understanding Task?","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2512.23213","citing_title":"Scoring, Reasoning, and Selecting the Best! Ensembling Large Language Models via a Peer-Review Process","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2412.05579","citing_title":"LLMs-as-Judges: A Comprehensive Survey on LLM-based Evaluation Methods","ref_index":39,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XQBEB6CG44IJMDMTFLPYJVTN5Y","json":"https://pith.science/pith/XQBEB6CG44IJMDMTFLPYJVTN5Y.json","graph_json":"https://pith.science/api/pith-number/XQBEB6CG44IJMDMTFLPYJVTN5Y/graph.json","events_json":"https://pith.science/api/pith-number/XQBEB6CG44IJMDMTFLPYJVTN5Y/events.json","paper":"https://pith.science/paper/XQBEB6CG"},"agent_actions":{"view_html":"https://pith.science/pith/XQBEB6CG44IJMDMTFLPYJVTN5Y","download_json":"https://pith.science/pith/XQBEB6CG44IJMDMTFLPYJVTN5Y.json","view_paper":"https://pith.science/paper/XQBEB6CG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.15641&json=true","fetch_graph":"https://pith.science/api/pith-number/XQBEB6CG44IJMDMTFLPYJVTN5Y/graph.json","fetch_events":"https://pith.science/api/pith-number/XQBEB6CG44IJMDMTFLPYJVTN5Y/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XQBEB6CG44IJMDMTFLPYJVTN5Y/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XQBEB6CG44IJMDMTFLPYJVTN5Y/action/storage_attestation","attest_author":"https://pith.science/pith/XQBEB6CG44IJMDMTFLPYJVTN5Y/action/author_attestation","sign_citation":"https://pith.science/pith/XQBEB6CG44IJMDMTFLPYJVTN5Y/action/citation_signature","submit_replication":"https://pith.science/pith/XQBEB6CG44IJMDMTFLPYJVTN5Y/action/replication_record"}},"created_at":"2026-07-05T08:26:18.208833+00:00","updated_at":"2026-07-05T08:26:18.208833+00:00"}