{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:OQH27T57TURF5DHIJOOKJCOPGT","short_pith_number":"pith:OQH27T57","schema_version":"1.0","canonical_sha256":"740fafcfbf9d225e8ce84b9ca489cf34eb39c427a6ed715a06fd21d602a2ad3e","source":{"kind":"arxiv","id":"2406.10229","version":1},"attestation_state":"computed","paper":{"title":"Quantifying Variance in Evaluation Benchmarks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Aaditya K. Singh, Andrew Poulton, Dieuwke Hupkes, Lovish Madaan, Pontus Stenetorp, Rylan Schaeffer, Sanmi Koyejo, Sharan Narang","submitted_at":"2024-06-14T17:59:54Z","abstract_excerpt":"Evaluation benchmarks are the cornerstone of measuring capabilities of large language models (LLMs), as well as driving progress in said capabilities. Originally designed to make claims about capabilities (or lack thereof) in fully pretrained models, evaluation benchmarks are now also extensively used to decide between various training choices. Despite this widespread usage, we rarely quantify the variance in our evaluation benchmarks, which dictates whether differences in performance are meaningful. Here, we define and measure a range of metrics geared towards measuring variance in evaluation"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.10229","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-06-14T17:59:54Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"fc5b9d15324b5e2c7dddb29436a9567e8d119c640e86fc416ae93d431bcabc15","abstract_canon_sha256":"f025ff8324adb33f35d6ad213a84f12d4458c56317acf6f520122dc4bf67a961"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:32:07.082359Z","signature_b64":"3f8XNAwTgTIO+reBlUpz25V7sSihTOerMn688uD0YESWN95FnmilvJKFoPHhQWisOrfiWKbMJfp0V8GMf+vWAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"740fafcfbf9d225e8ce84b9ca489cf34eb39c427a6ed715a06fd21d602a2ad3e","last_reissued_at":"2026-07-05T08:32:07.081847Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:32:07.081847Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Quantifying Variance in Evaluation Benchmarks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Aaditya K. Singh, Andrew Poulton, Dieuwke Hupkes, Lovish Madaan, Pontus Stenetorp, Rylan Schaeffer, Sanmi Koyejo, Sharan Narang","submitted_at":"2024-06-14T17:59:54Z","abstract_excerpt":"Evaluation benchmarks are the cornerstone of measuring capabilities of large language models (LLMs), as well as driving progress in said capabilities. Originally designed to make claims about capabilities (or lack thereof) in fully pretrained models, evaluation benchmarks are now also extensively used to decide between various training choices. Despite this widespread usage, we rarely quantify the variance in our evaluation benchmarks, which dictates whether differences in performance are meaningful. Here, we define and measure a range of metrics geared towards measuring variance in evaluation"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.10229","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.10229/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.10229","created_at":"2026-07-05T08:32:07.081912+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.10229v1","created_at":"2026-07-05T08:32:07.081912+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.10229","created_at":"2026-07-05T08:32:07.081912+00:00"},{"alias_kind":"pith_short_12","alias_value":"OQH27T57TURF","created_at":"2026-07-05T08:32:07.081912+00:00"},{"alias_kind":"pith_short_16","alias_value":"OQH27T57TURF5DHI","created_at":"2026-07-05T08:32:07.081912+00:00"},{"alias_kind":"pith_short_8","alias_value":"OQH27T57","created_at":"2026-07-05T08:32:07.081912+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07946","citing_title":"DeepSWE: Measuring Frontier Coding Agents on Original, Long-Horizon Engineering Tasks","ref_index":8,"is_internal_anchor":true},{"citing_arxiv_id":"2606.17426","citing_title":"Bounded Difference Concentration for Infinitely Exchangeable Sequences with Applications to AI Benchmark Uncertainty","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28551","citing_title":"DataComp-VLM: Improved Open Datasets for Vision-Language Models","ref_index":197,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05029","citing_title":"Validity Threats for Foundation Model Research","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28551","citing_title":"DataComp-VLM: Improved Open Datasets for Vision-Language Models","ref_index":197,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28190","citing_title":"The Harder Text Embedding Benchmark (HTEB): Beyond One-dimensional Static Robustness","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30315","citing_title":"Resolution Diagnostics for Paired LLM Evaluation","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2502.12120","citing_title":"LLMs on the Line: Data Determines Loss-to-Loss Scaling Laws","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2503.17181","citing_title":"A Study of LLMs' Preferences for Libraries and Programming Languages","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2510.13786","citing_title":"The Art of Scaling Reinforcement Learning Compute for LLMs","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23114","citing_title":"A Tale of Two Variances: When Single-Seed Benchmarks Fail in Bayesian Deep Learning","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06213","citing_title":"Beyond Fixed Benchmarks and Worst-Case Attacks: Dynamic Boundary Evaluation for Language Models","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OQH27T57TURF5DHIJOOKJCOPGT","json":"https://pith.science/pith/OQH27T57TURF5DHIJOOKJCOPGT.json","graph_json":"https://pith.science/api/pith-number/OQH27T57TURF5DHIJOOKJCOPGT/graph.json","events_json":"https://pith.science/api/pith-number/OQH27T57TURF5DHIJOOKJCOPGT/events.json","paper":"https://pith.science/paper/OQH27T57"},"agent_actions":{"view_html":"https://pith.science/pith/OQH27T57TURF5DHIJOOKJCOPGT","download_json":"https://pith.science/pith/OQH27T57TURF5DHIJOOKJCOPGT.json","view_paper":"https://pith.science/paper/OQH27T57","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.10229&json=true","fetch_graph":"https://pith.science/api/pith-number/OQH27T57TURF5DHIJOOKJCOPGT/graph.json","fetch_events":"https://pith.science/api/pith-number/OQH27T57TURF5DHIJOOKJCOPGT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OQH27T57TURF5DHIJOOKJCOPGT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OQH27T57TURF5DHIJOOKJCOPGT/action/storage_attestation","attest_author":"https://pith.science/pith/OQH27T57TURF5DHIJOOKJCOPGT/action/author_attestation","sign_citation":"https://pith.science/pith/OQH27T57TURF5DHIJOOKJCOPGT/action/citation_signature","submit_replication":"https://pith.science/pith/OQH27T57TURF5DHIJOOKJCOPGT/action/replication_record"}},"created_at":"2026-07-05T08:32:07.081912+00:00","updated_at":"2026-07-05T08:32:07.081912+00:00"}