{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:DM75YXPBD6VOFIYIVH7QRWFQ7Q","short_pith_number":"pith:DM75YXPB","schema_version":"1.0","canonical_sha256":"1b3fdc5de11faae2a308a9ff08d8b0fc33d4281a48846d662a380ab9d46c8907","source":{"kind":"arxiv","id":"2410.12265","version":2},"attestation_state":"computed","paper":{"title":"Auto-PRE: An Automatic and Cost-Efficient Peer-Review Framework for Language Generation Evaluation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dingbo Yuan, Haitao Li, Junjie Chen, Jun Zhou, Min Zhang, Qingyao Ai, Shaoping Ma, Weihang Su, Xudong Wang, Yiqun Liu, Yujia Zhou, Zhumin Chu","submitted_at":"2024-10-16T06:06:06Z","abstract_excerpt":"The rapid development of large language models (LLMs) has highlighted the need for efficient and reliable methods to evaluate their performance. Traditional evaluation methods often face challenges like high costs, limited task formats, dependence on human references, and systematic biases. To address these limitations, we propose Auto-PRE, an automatic LLM evaluation framework inspired by the peer review process. Unlike previous approaches that rely on human annotations, Auto-PRE automatically selects evaluator LLMs based on three core traits: consistency, pertinence, and self-confidence, whi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.12265","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-10-16T06:06:06Z","cross_cats_sorted":[],"title_canon_sha256":"9bd1af8bf30c7bf1cecf77b36b6569d40ca35aee647be4b590944e21d05701fe","abstract_canon_sha256":"bc9e02170a0774924be1f173805acd922a77660f0f7624c6b0874dad70ac648c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-11T02:18:11.060426Z","signature_b64":"1ZfQ5Vka39mf2HMZGyAkx+dITqtyVZfqr3s2+TlOypr2hAA90qSdPTbheAOa3AWLTq+yIPa8WVdHRvZ99rvABg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1b3fdc5de11faae2a308a9ff08d8b0fc33d4281a48846d662a380ab9d46c8907","last_reissued_at":"2026-08-11T02:18:11.058581Z","signature_status":"signed_v1","first_computed_at":"2026-08-11T02:18:11.058581Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Auto-PRE: An Automatic and Cost-Efficient Peer-Review Framework for Language Generation Evaluation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dingbo Yuan, Haitao Li, Junjie Chen, Jun Zhou, Min Zhang, Qingyao Ai, Shaoping Ma, Weihang Su, Xudong Wang, Yiqun Liu, Yujia Zhou, Zhumin Chu","submitted_at":"2024-10-16T06:06:06Z","abstract_excerpt":"The rapid development of large language models (LLMs) has highlighted the need for efficient and reliable methods to evaluate their performance. Traditional evaluation methods often face challenges like high costs, limited task formats, dependence on human references, and systematic biases. To address these limitations, we propose Auto-PRE, an automatic LLM evaluation framework inspired by the peer review process. Unlike previous approaches that rely on human annotations, Auto-PRE automatically selects evaluator LLMs based on three core traits: consistency, pertinence, and self-confidence, whi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.12265","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.12265/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.12265","created_at":"2026-08-11T02:18:11.059336+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.12265v2","created_at":"2026-08-11T02:18:11.059336+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.12265","created_at":"2026-08-11T02:18:11.059336+00:00"},{"alias_kind":"pith_short_12","alias_value":"DM75YXPBD6VO","created_at":"2026-08-11T02:18:11.059336+00:00"},{"alias_kind":"pith_short_16","alias_value":"DM75YXPBD6VOFIYI","created_at":"2026-08-11T02:18:11.059336+00:00"},{"alias_kind":"pith_short_8","alias_value":"DM75YXPB","created_at":"2026-08-11T02:18:11.059336+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2512.23213","citing_title":"Scoring, Reasoning, and Selecting the Best! Ensembling Large Language Models via a Peer-Review Process","ref_index":5,"is_internal_anchor":true},{"citing_arxiv_id":"2412.05579","citing_title":"LLMs-as-Judges: A Comprehensive Survey on LLM-based Evaluation Methods","ref_index":27,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DM75YXPBD6VOFIYIVH7QRWFQ7Q","json":"https://pith.science/pith/DM75YXPBD6VOFIYIVH7QRWFQ7Q.json","graph_json":"https://pith.science/api/pith-number/DM75YXPBD6VOFIYIVH7QRWFQ7Q/graph.json","events_json":"https://pith.science/api/pith-number/DM75YXPBD6VOFIYIVH7QRWFQ7Q/events.json","paper":"https://pith.science/paper/DM75YXPB"},"agent_actions":{"view_html":"https://pith.science/pith/DM75YXPBD6VOFIYIVH7QRWFQ7Q","download_json":"https://pith.science/pith/DM75YXPBD6VOFIYIVH7QRWFQ7Q.json","view_paper":"https://pith.science/paper/DM75YXPB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.12265&json=true","fetch_graph":"https://pith.science/api/pith-number/DM75YXPBD6VOFIYIVH7QRWFQ7Q/graph.json","fetch_events":"https://pith.science/api/pith-number/DM75YXPBD6VOFIYIVH7QRWFQ7Q/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DM75YXPBD6VOFIYIVH7QRWFQ7Q/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DM75YXPBD6VOFIYIVH7QRWFQ7Q/action/storage_attestation","attest_author":"https://pith.science/pith/DM75YXPBD6VOFIYIVH7QRWFQ7Q/action/author_attestation","sign_citation":"https://pith.science/pith/DM75YXPBD6VOFIYIVH7QRWFQ7Q/action/citation_signature","submit_replication":"https://pith.science/pith/DM75YXPBD6VOFIYIVH7QRWFQ7Q/action/replication_record"}},"created_at":"2026-08-11T02:18:11.059336+00:00","updated_at":"2026-08-11T02:18:11.059336+00:00"}