{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:LTLUJA5AKZEBLBAVMYMVMOTQE7","short_pith_number":"pith:LTLUJA5A","schema_version":"1.0","canonical_sha256":"5cd74483a056481584156619563a7027e3b87087edf5f4f381e47df6685bf138","source":{"kind":"arxiv","id":"2308.11696","version":5},"attestation_state":"computed","paper":{"title":"Efficient Benchmarking of Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.LG"],"primary_cat":"cs.CL","authors_text":"Ariel Gera, Elron Bandel, Eyal Shnarch, Leshem Choshen, Liat Ein-Dor, Michal Shmueli-Scheuer, Noam Slonim, Ofir Arviv, Yotam Perlitz","submitted_at":"2023-08-22T17:59:30Z","abstract_excerpt":"The increasing versatility of language models (LMs) has given rise to a new class of benchmarks that comprehensively assess a broad range of capabilities. Such benchmarks are associated with massive computational costs, extending to thousands of GPU hours per model. However, the efficiency aspect of these evaluation efforts had raised little discussion in the literature. In this work, we present the problem of Efficient Benchmarking, namely, intelligently reducing the computation costs of LM evaluation without compromising reliability. Using the HELM benchmark as a test case, we investigate ho"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.11696","kind":"arxiv","version":5},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-08-22T17:59:30Z","cross_cats_sorted":["cs.AI","cs.CV","cs.LG"],"title_canon_sha256":"58eb45aec2dc0c6cd9ed6278f6da701c1b1216505c103c49b7a6a4c61f7d7531","abstract_canon_sha256":"28a9e73830da07fb32849c93cc677b49e7a248a10ebbbff0d32471e2bb7e60fd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:02:45.669699Z","signature_b64":"l5byeCSxPtRJScUeiabtgGm65BsuUFP+y0YuL1ZySkyb7INDhKkj+feGhHGgB/aiZ/GNOo9mDYb/RaWCmyAYDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5cd74483a056481584156619563a7027e3b87087edf5f4f381e47df6685bf138","last_reissued_at":"2026-07-05T08:02:45.669180Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:02:45.669180Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Efficient Benchmarking of Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.LG"],"primary_cat":"cs.CL","authors_text":"Ariel Gera, Elron Bandel, Eyal Shnarch, Leshem Choshen, Liat Ein-Dor, Michal Shmueli-Scheuer, Noam Slonim, Ofir Arviv, Yotam Perlitz","submitted_at":"2023-08-22T17:59:30Z","abstract_excerpt":"The increasing versatility of language models (LMs) has given rise to a new class of benchmarks that comprehensively assess a broad range of capabilities. Such benchmarks are associated with massive computational costs, extending to thousands of GPU hours per model. However, the efficiency aspect of these evaluation efforts had raised little discussion in the literature. In this work, we present the problem of Efficient Benchmarking, namely, intelligently reducing the computation costs of LM evaluation without compromising reliability. Using the HELM benchmark as a test case, we investigate ho"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.11696","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.11696/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.11696","created_at":"2026-07-05T08:02:45.669239+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.11696v5","created_at":"2026-07-05T08:02:45.669239+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.11696","created_at":"2026-07-05T08:02:45.669239+00:00"},{"alias_kind":"pith_short_12","alias_value":"LTLUJA5AKZEB","created_at":"2026-07-05T08:02:45.669239+00:00"},{"alias_kind":"pith_short_16","alias_value":"LTLUJA5AKZEBLBAV","created_at":"2026-07-05T08:02:45.669239+00:00"},{"alias_kind":"pith_short_8","alias_value":"LTLUJA5A","created_at":"2026-07-05T08:02:45.669239+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.06941","citing_title":"Quantum-Inspired Trace-Augmented Evidence Selection for Reasoning over Structured Hypothesis Spaces","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30916","citing_title":"Welfare, Improvability, and Variance: A Principal-Agent Approach to Optimal Benchmark Item Aggregation","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01400","citing_title":"Consistent and Distinctive: LLM Benchmark Efficiency via Maximum Independent Set Prompt Selection on Similarity Graphs","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2404.18923","citing_title":"Holmes: A Benchmark to Assess the Linguistic Competence of Language Models","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07096","citing_title":"Query-efficient model evaluation using cached responses","ref_index":112,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LTLUJA5AKZEBLBAVMYMVMOTQE7","json":"https://pith.science/pith/LTLUJA5AKZEBLBAVMYMVMOTQE7.json","graph_json":"https://pith.science/api/pith-number/LTLUJA5AKZEBLBAVMYMVMOTQE7/graph.json","events_json":"https://pith.science/api/pith-number/LTLUJA5AKZEBLBAVMYMVMOTQE7/events.json","paper":"https://pith.science/paper/LTLUJA5A"},"agent_actions":{"view_html":"https://pith.science/pith/LTLUJA5AKZEBLBAVMYMVMOTQE7","download_json":"https://pith.science/pith/LTLUJA5AKZEBLBAVMYMVMOTQE7.json","view_paper":"https://pith.science/paper/LTLUJA5A","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.11696&json=true","fetch_graph":"https://pith.science/api/pith-number/LTLUJA5AKZEBLBAVMYMVMOTQE7/graph.json","fetch_events":"https://pith.science/api/pith-number/LTLUJA5AKZEBLBAVMYMVMOTQE7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LTLUJA5AKZEBLBAVMYMVMOTQE7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LTLUJA5AKZEBLBAVMYMVMOTQE7/action/storage_attestation","attest_author":"https://pith.science/pith/LTLUJA5AKZEBLBAVMYMVMOTQE7/action/author_attestation","sign_citation":"https://pith.science/pith/LTLUJA5AKZEBLBAVMYMVMOTQE7/action/citation_signature","submit_replication":"https://pith.science/pith/LTLUJA5AKZEBLBAVMYMVMOTQE7/action/replication_record"}},"created_at":"2026-07-05T08:02:45.669239+00:00","updated_at":"2026-07-05T08:02:45.669239+00:00"}