{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:4XPFWIKY77C5MD5N7F4GKURWBE","short_pith_number":"pith:4XPFWIKY","schema_version":"1.0","canonical_sha256":"e5de5b2158ffc5d60fadf978655236090768532d16be9e382be9827568f16653","source":{"kind":"arxiv","id":"2503.01747","version":3},"attestation_state":"computed","paper":{"title":"Position: Don't Use the CLT in LLM Evals With Fewer Than a Few Hundred Datapoints","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG","stat.ML"],"primary_cat":"cs.AI","authors_text":"Desi R. Ivanova, Laurence Aitchison, Sam Bowyer","submitted_at":"2025-03-03T17:15:17Z","abstract_excerpt":"Rigorous statistical evaluations of large language models (LLMs), including valid error bars and significance testing, are essential for meaningful and reliable performance assessment. Currently, when such statistical measures are reported, they typically rely on the Central Limit Theorem (CLT). In this position paper, we argue that while CLT-based methods for uncertainty quantification are appropriate when benchmarks consist of thousands of examples, they fail to provide adequate uncertainty estimates for LLM evaluations that rely on smaller, highly specialized benchmarks. In these small-data"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.01747","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-03-03T17:15:17Z","cross_cats_sorted":["cs.LG","stat.ML"],"title_canon_sha256":"07f0a7a246295f78b8004bc4b1a65ba5518f4a5d5c70aebe0ce40818ffec4e0d","abstract_canon_sha256":"8605c40c55e8a3569fa064e5c937180c5a2dce2abbc946926f16a231a5d427a3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:11:21.825036Z","signature_b64":"OVuYy43971nFwMW/egDBQaMpyan6kpz7LeSzA62VuDrNmgGDW7rzrIj5pwTCqcso7LQvn9ewNR5UNsLECxDADA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e5de5b2158ffc5d60fadf978655236090768532d16be9e382be9827568f16653","last_reissued_at":"2026-07-05T11:11:21.824581Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:11:21.824581Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Position: Don't Use the CLT in LLM Evals With Fewer Than a Few Hundred Datapoints","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG","stat.ML"],"primary_cat":"cs.AI","authors_text":"Desi R. Ivanova, Laurence Aitchison, Sam Bowyer","submitted_at":"2025-03-03T17:15:17Z","abstract_excerpt":"Rigorous statistical evaluations of large language models (LLMs), including valid error bars and significance testing, are essential for meaningful and reliable performance assessment. Currently, when such statistical measures are reported, they typically rely on the Central Limit Theorem (CLT). In this position paper, we argue that while CLT-based methods for uncertainty quantification are appropriate when benchmarks consist of thousands of examples, they fail to provide adequate uncertainty estimates for LLM evaluations that rely on smaller, highly specialized benchmarks. In these small-data"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.01747","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.01747/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.01747","created_at":"2026-07-05T11:11:21.824638+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.01747v3","created_at":"2026-07-05T11:11:21.824638+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.01747","created_at":"2026-07-05T11:11:21.824638+00:00"},{"alias_kind":"pith_short_12","alias_value":"4XPFWIKY77C5","created_at":"2026-07-05T11:11:21.824638+00:00"},{"alias_kind":"pith_short_16","alias_value":"4XPFWIKY77C5MD5N","created_at":"2026-07-05T11:11:21.824638+00:00"},{"alias_kind":"pith_short_8","alias_value":"4XPFWIKY","created_at":"2026-07-05T11:11:21.824638+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20963","citing_title":"Generative Responsible AI Data Evaluation Schema (GRAIDES) for AI Assurance in Local Government","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28324","citing_title":"The multiply iterated law of the iterated logarithm: game-theoretic foundations of sequential detection boundaries","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28324","citing_title":"The multiply iterated law of the iterated logarithm: game-theoretic foundations of sequential detection boundaries","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2510.04265","citing_title":"Don't Pass@k: A Bayesian Framework for Large Language Model Evaluation","ref_index":46,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4XPFWIKY77C5MD5N7F4GKURWBE","json":"https://pith.science/pith/4XPFWIKY77C5MD5N7F4GKURWBE.json","graph_json":"https://pith.science/api/pith-number/4XPFWIKY77C5MD5N7F4GKURWBE/graph.json","events_json":"https://pith.science/api/pith-number/4XPFWIKY77C5MD5N7F4GKURWBE/events.json","paper":"https://pith.science/paper/4XPFWIKY"},"agent_actions":{"view_html":"https://pith.science/pith/4XPFWIKY77C5MD5N7F4GKURWBE","download_json":"https://pith.science/pith/4XPFWIKY77C5MD5N7F4GKURWBE.json","view_paper":"https://pith.science/paper/4XPFWIKY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.01747&json=true","fetch_graph":"https://pith.science/api/pith-number/4XPFWIKY77C5MD5N7F4GKURWBE/graph.json","fetch_events":"https://pith.science/api/pith-number/4XPFWIKY77C5MD5N7F4GKURWBE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4XPFWIKY77C5MD5N7F4GKURWBE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4XPFWIKY77C5MD5N7F4GKURWBE/action/storage_attestation","attest_author":"https://pith.science/pith/4XPFWIKY77C5MD5N7F4GKURWBE/action/author_attestation","sign_citation":"https://pith.science/pith/4XPFWIKY77C5MD5N7F4GKURWBE/action/citation_signature","submit_replication":"https://pith.science/pith/4XPFWIKY77C5MD5N7F4GKURWBE/action/replication_record"}},"created_at":"2026-07-05T11:11:21.824638+00:00","updated_at":"2026-07-05T11:11:21.824638+00:00"}