{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:O7QEOJCFKKWWMRI3PC6PMNU44Q","short_pith_number":"pith:O7QEOJCF","schema_version":"1.0","canonical_sha256":"77e047244552ad66451b78bcf6369ce4131f15b3d1bad805e53fe08745434fc8","source":{"kind":"arxiv","id":"2407.12844","version":2},"attestation_state":"computed","paper":{"title":"metabench -- A Sparse Benchmark of Reasoning and Knowledge in Large Language Models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.LG","stat.ML"],"primary_cat":"cs.CL","authors_text":"Alex Kipnis, Eric Schulz, Konstantinos Voudouris, Luca M. Schulze Buschoff","submitted_at":"2024-07-04T17:57:38Z","abstract_excerpt":"Large Language Models (LLMs) vary in their abilities on a range of tasks. Initiatives such as the Open LLM Leaderboard aim to quantify these differences with several large benchmarks (sets of test items to which an LLM can respond either correctly or incorrectly). However, high correlations within and between benchmark scores suggest that (1) there exists a small set of common underlying abilities that these benchmarks measure, and (2) items tap into redundant information and the benchmarks may thus be considerably compressed. We use data from n > 5000 LLMs to identify the most informative ite"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.12844","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-07-04T17:57:38Z","cross_cats_sorted":["cs.LG","stat.ML"],"title_canon_sha256":"e799fbe20cbe362ceb9df04ced5e08dc6ae14ad94e00241d65ea512c2d2a8c29","abstract_canon_sha256":"3d4e8105c85b49a50f9af6dd65753b6a4d229e8a12edcea22843fee1e920b288"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:17:09.740576Z","signature_b64":"ePxgNN63Ly14v2F1VFLegNh0FvNXReN0huuQPIyzHozPZhUw4p4sPCHjB4n/ZNrvUcOvHXXK/Vh1pQAh5vzICA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"77e047244552ad66451b78bcf6369ce4131f15b3d1bad805e53fe08745434fc8","last_reissued_at":"2026-07-05T10:17:09.740067Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:17:09.740067Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"metabench -- A Sparse Benchmark of Reasoning and Knowledge in Large Language Models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.LG","stat.ML"],"primary_cat":"cs.CL","authors_text":"Alex Kipnis, Eric Schulz, Konstantinos Voudouris, Luca M. Schulze Buschoff","submitted_at":"2024-07-04T17:57:38Z","abstract_excerpt":"Large Language Models (LLMs) vary in their abilities on a range of tasks. Initiatives such as the Open LLM Leaderboard aim to quantify these differences with several large benchmarks (sets of test items to which an LLM can respond either correctly or incorrectly). However, high correlations within and between benchmark scores suggest that (1) there exists a small set of common underlying abilities that these benchmarks measure, and (2) items tap into redundant information and the benchmarks may thus be considerably compressed. We use data from n > 5000 LLMs to identify the most informative ite"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.12844","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.12844/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.12844","created_at":"2026-07-05T10:17:09.740123+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.12844v2","created_at":"2026-07-05T10:17:09.740123+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.12844","created_at":"2026-07-05T10:17:09.740123+00:00"},{"alias_kind":"pith_short_12","alias_value":"O7QEOJCFKKWW","created_at":"2026-07-05T10:17:09.740123+00:00"},{"alias_kind":"pith_short_16","alias_value":"O7QEOJCFKKWWMRI3","created_at":"2026-07-05T10:17:09.740123+00:00"},{"alias_kind":"pith_short_8","alias_value":"O7QEOJCF","created_at":"2026-07-05T10:17:09.740123+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01152","citing_title":"AGC-Bench: Measuring Artificial General Creativity","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2606.14516","citing_title":"Every Eval Ever: A Unifying Schema and Community Repository for AI Evaluation Results","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01152","citing_title":"AGC-Bench: Measuring Artificial General Creativity","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07616","citing_title":"Item Response Scaling Laws: A Measurement Theory Approach for Efficient and Generalizable Neural Scaling Estimation","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2507.23009","citing_title":"Position: Stop Evaluating AI with Human Tests, Develop Principled, AI-specific Tests instead","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06213","citing_title":"Beyond Fixed Benchmarks and Worst-Case Attacks: Dynamic Boundary Evaluation for Language Models","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11328","citing_title":"Select Smarter, Not More: Prompt-Aware Evaluation Scheduling with Submodular Guarantees","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12843","citing_title":"Growing Pains: Extensible and Efficient LLM Benchmarking Via Fixed Parameter Calibration","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/O7QEOJCFKKWWMRI3PC6PMNU44Q","json":"https://pith.science/pith/O7QEOJCFKKWWMRI3PC6PMNU44Q.json","graph_json":"https://pith.science/api/pith-number/O7QEOJCFKKWWMRI3PC6PMNU44Q/graph.json","events_json":"https://pith.science/api/pith-number/O7QEOJCFKKWWMRI3PC6PMNU44Q/events.json","paper":"https://pith.science/paper/O7QEOJCF"},"agent_actions":{"view_html":"https://pith.science/pith/O7QEOJCFKKWWMRI3PC6PMNU44Q","download_json":"https://pith.science/pith/O7QEOJCFKKWWMRI3PC6PMNU44Q.json","view_paper":"https://pith.science/paper/O7QEOJCF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.12844&json=true","fetch_graph":"https://pith.science/api/pith-number/O7QEOJCFKKWWMRI3PC6PMNU44Q/graph.json","fetch_events":"https://pith.science/api/pith-number/O7QEOJCFKKWWMRI3PC6PMNU44Q/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/O7QEOJCFKKWWMRI3PC6PMNU44Q/action/timestamp_anchor","attest_storage":"https://pith.science/pith/O7QEOJCFKKWWMRI3PC6PMNU44Q/action/storage_attestation","attest_author":"https://pith.science/pith/O7QEOJCFKKWWMRI3PC6PMNU44Q/action/author_attestation","sign_citation":"https://pith.science/pith/O7QEOJCFKKWWMRI3PC6PMNU44Q/action/citation_signature","submit_replication":"https://pith.science/pith/O7QEOJCFKKWWMRI3PC6PMNU44Q/action/replication_record"}},"created_at":"2026-07-05T10:17:09.740123+00:00","updated_at":"2026-07-05T10:17:09.740123+00:00"}