{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:4JKIHM35IL6S4E5X2DQAAIXDTZ","short_pith_number":"pith:4JKIHM35","schema_version":"1.0","canonical_sha256":"e25483b37d42fd2e13b7d0e00022e39e42b4161e5351f3f45ffc90c9568ddf10","source":{"kind":"arxiv","id":"2506.07673","version":1},"attestation_state":"computed","paper":{"title":"How Benchmark Prediction from Fewer Data Misses the Mark","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Florian E. Dorner, Guanhua Zhang, Moritz Hardt","submitted_at":"2025-06-09T11:50:41Z","abstract_excerpt":"Large language model (LLM) evaluation is increasingly costly, prompting interest in methods that speed up evaluation by shrinking benchmark datasets. Benchmark prediction (also called efficient LLM evaluation) aims to select a small subset of evaluation points and predict overall benchmark performance from that subset. In this paper, we systematically assess the strengths and limitations of 11 benchmark prediction methods across 19 diverse benchmarks. First, we identify a highly competitive baseline: Take a random sample and fit a regression model on the sample to predict missing entries. Outp"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.07673","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-09T11:50:41Z","cross_cats_sorted":[],"title_canon_sha256":"7296b5794cc9adfcb55d49347aeffdd717fff98dedbc64147ff28426ea4182fd","abstract_canon_sha256":"06844c55fcb190ce755cb680f47bf1e64758e9ccedaeb41403a149fee04e6520"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:18:31.339768Z","signature_b64":"xsNhbjnqd4KyAUQSlW8Fv97NGZRJuaCpI8kP77acimqnSFPuZ0jtZ9jx1qXMibcZBUsAczWVeOWEAlM3ZwuUCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e25483b37d42fd2e13b7d0e00022e39e42b4161e5351f3f45ffc90c9568ddf10","last_reissued_at":"2026-07-05T11:18:31.339269Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:18:31.339269Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"How Benchmark Prediction from Fewer Data Misses the Mark","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Florian E. Dorner, Guanhua Zhang, Moritz Hardt","submitted_at":"2025-06-09T11:50:41Z","abstract_excerpt":"Large language model (LLM) evaluation is increasingly costly, prompting interest in methods that speed up evaluation by shrinking benchmark datasets. Benchmark prediction (also called efficient LLM evaluation) aims to select a small subset of evaluation points and predict overall benchmark performance from that subset. In this paper, we systematically assess the strengths and limitations of 11 benchmark prediction methods across 19 diverse benchmarks. First, we identify a highly competitive baseline: Take a random sample and fit a regression model on the sample to predict missing entries. Outp"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.07673","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.07673/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.07673","created_at":"2026-07-05T11:18:31.339327+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.07673v1","created_at":"2026-07-05T11:18:31.339327+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.07673","created_at":"2026-07-05T11:18:31.339327+00:00"},{"alias_kind":"pith_short_12","alias_value":"4JKIHM35IL6S","created_at":"2026-07-05T11:18:31.339327+00:00"},{"alias_kind":"pith_short_16","alias_value":"4JKIHM35IL6S4E5X","created_at":"2026-07-05T11:18:31.339327+00:00"},{"alias_kind":"pith_short_8","alias_value":"4JKIHM35","created_at":"2026-07-05T11:18:31.339327+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":5,"sample":[{"citing_arxiv_id":"2606.24020","citing_title":"You Don't Need to Run Every Eval","ref_index":6,"is_internal_anchor":true},{"citing_arxiv_id":"2606.05029","citing_title":"Validity Threats for Foundation Model Research","ref_index":110,"is_internal_anchor":true},{"citing_arxiv_id":"2606.03330","citing_title":"FLIPS: Instance-Fingerprinting for LLMs via Pseudo-random Sequences","ref_index":60,"is_internal_anchor":true},{"citing_arxiv_id":"2601.20251","citing_title":"Efficient Evaluation of LLM Performance with Statistical Guarantees","ref_index":17,"is_internal_anchor":true},{"citing_arxiv_id":"2605.05973","citing_title":"Towards Reliable LLM Evaluation: Correcting the Winner's Curse in Adaptive Benchmarking","ref_index":32,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4JKIHM35IL6S4E5X2DQAAIXDTZ","json":"https://pith.science/pith/4JKIHM35IL6S4E5X2DQAAIXDTZ.json","graph_json":"https://pith.science/api/pith-number/4JKIHM35IL6S4E5X2DQAAIXDTZ/graph.json","events_json":"https://pith.science/api/pith-number/4JKIHM35IL6S4E5X2DQAAIXDTZ/events.json","paper":"https://pith.science/paper/4JKIHM35"},"agent_actions":{"view_html":"https://pith.science/pith/4JKIHM35IL6S4E5X2DQAAIXDTZ","download_json":"https://pith.science/pith/4JKIHM35IL6S4E5X2DQAAIXDTZ.json","view_paper":"https://pith.science/paper/4JKIHM35","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.07673&json=true","fetch_graph":"https://pith.science/api/pith-number/4JKIHM35IL6S4E5X2DQAAIXDTZ/graph.json","fetch_events":"https://pith.science/api/pith-number/4JKIHM35IL6S4E5X2DQAAIXDTZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4JKIHM35IL6S4E5X2DQAAIXDTZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4JKIHM35IL6S4E5X2DQAAIXDTZ/action/storage_attestation","attest_author":"https://pith.science/pith/4JKIHM35IL6S4E5X2DQAAIXDTZ/action/author_attestation","sign_citation":"https://pith.science/pith/4JKIHM35IL6S4E5X2DQAAIXDTZ/action/citation_signature","submit_replication":"https://pith.science/pith/4JKIHM35IL6S4E5X2DQAAIXDTZ/action/replication_record"}},"created_at":"2026-07-05T11:18:31.339327+00:00","updated_at":"2026-07-05T11:18:31.339327+00:00"}