{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ZV6PMYKEWQ2ERB3IMZLJ4FKD26","short_pith_number":"pith:ZV6PMYKE","schema_version":"1.0","canonical_sha256":"cd7cf66144b43448876866569e1543d78415d379f31e6fa7a663b0c2e71ec673","source":{"kind":"arxiv","id":"2502.01683","version":1},"attestation_state":"computed","paper":{"title":"LLM-Powered Benchmark Factory: Reliable, Generic, and Efficient","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Boyuan Pan, Chuyi Tan, Jiayi Shi, Kan Li, Peiwen Yuan, Shaoxiong Feng, Xinglin Wang, Yao Hu, Yiwei Li, Yueqi Zhang","submitted_at":"2025-02-02T06:36:01Z","abstract_excerpt":"The rapid advancement of large language models (LLMs) has led to a surge in both model supply and application demands. To facilitate effective matching between them, reliable, generic and efficient benchmark generators are widely needed. However, human annotators are constrained by inefficiency, and current LLM benchmark generators not only lack generalizability but also struggle with limited reliability, as they lack a comprehensive evaluation framework for validation and optimization. To fill this gap, we first propose an automated and unbiased evaluation framework, structured around four di"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.01683","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-02-02T06:36:01Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"1db39a16fb54fbb137eaf590516d2f6fb0ca4c2daf4a5d2345831eec8bf969ab","abstract_canon_sha256":"e37095abca0ec1ef5860497552553757478f88db66a5e794c9ef09ddb71498f0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:09:17.626515Z","signature_b64":"bQPfzp/UU/Y0YZzQBcPbr/ED86cs0ppaizV8Vj25/Bdn0nRCW5+7p0nPM7E40PC2699uztRG/97AQcSDCKiFDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cd7cf66144b43448876866569e1543d78415d379f31e6fa7a663b0c2e71ec673","last_reissued_at":"2026-07-05T10:09:17.625988Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:09:17.625988Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LLM-Powered Benchmark Factory: Reliable, Generic, and Efficient","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Boyuan Pan, Chuyi Tan, Jiayi Shi, Kan Li, Peiwen Yuan, Shaoxiong Feng, Xinglin Wang, Yao Hu, Yiwei Li, Yueqi Zhang","submitted_at":"2025-02-02T06:36:01Z","abstract_excerpt":"The rapid advancement of large language models (LLMs) has led to a surge in both model supply and application demands. To facilitate effective matching between them, reliable, generic and efficient benchmark generators are widely needed. However, human annotators are constrained by inefficiency, and current LLM benchmark generators not only lack generalizability but also struggle with limited reliability, as they lack a comprehensive evaluation framework for validation and optimization. To fill this gap, we first propose an automated and unbiased evaluation framework, structured around four di"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.01683","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.01683/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.01683","created_at":"2026-07-05T10:09:17.626057+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.01683v1","created_at":"2026-07-05T10:09:17.626057+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.01683","created_at":"2026-07-05T10:09:17.626057+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZV6PMYKEWQ2E","created_at":"2026-07-05T10:09:17.626057+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZV6PMYKEWQ2ERB3I","created_at":"2026-07-05T10:09:17.626057+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZV6PMYKE","created_at":"2026-07-05T10:09:17.626057+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.04386","citing_title":"Automatically Generating Hard Math Problems from Hypothesis-Driven Error Analysis","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZV6PMYKEWQ2ERB3IMZLJ4FKD26","json":"https://pith.science/pith/ZV6PMYKEWQ2ERB3IMZLJ4FKD26.json","graph_json":"https://pith.science/api/pith-number/ZV6PMYKEWQ2ERB3IMZLJ4FKD26/graph.json","events_json":"https://pith.science/api/pith-number/ZV6PMYKEWQ2ERB3IMZLJ4FKD26/events.json","paper":"https://pith.science/paper/ZV6PMYKE"},"agent_actions":{"view_html":"https://pith.science/pith/ZV6PMYKEWQ2ERB3IMZLJ4FKD26","download_json":"https://pith.science/pith/ZV6PMYKEWQ2ERB3IMZLJ4FKD26.json","view_paper":"https://pith.science/paper/ZV6PMYKE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.01683&json=true","fetch_graph":"https://pith.science/api/pith-number/ZV6PMYKEWQ2ERB3IMZLJ4FKD26/graph.json","fetch_events":"https://pith.science/api/pith-number/ZV6PMYKEWQ2ERB3IMZLJ4FKD26/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZV6PMYKEWQ2ERB3IMZLJ4FKD26/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZV6PMYKEWQ2ERB3IMZLJ4FKD26/action/storage_attestation","attest_author":"https://pith.science/pith/ZV6PMYKEWQ2ERB3IMZLJ4FKD26/action/author_attestation","sign_citation":"https://pith.science/pith/ZV6PMYKEWQ2ERB3IMZLJ4FKD26/action/citation_signature","submit_replication":"https://pith.science/pith/ZV6PMYKEWQ2ERB3IMZLJ4FKD26/action/replication_record"}},"created_at":"2026-07-05T10:09:17.626057+00:00","updated_at":"2026-07-05T10:09:17.626057+00:00"}