{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:MPUKCXH5EBVKFNZJJCPMS556MZ","short_pith_number":"pith:MPUKCXH5","schema_version":"1.0","canonical_sha256":"63e8a15cfd206aa2b729489ec977be667a1222e07a30ad7ccc7b4577dd602e6d","source":{"kind":"arxiv","id":"2412.15194","version":1},"attestation_state":"computed","paper":{"title":"MMLU-CF: A Contamination-free Multi-task Language Understanding Benchmark","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.PF"],"primary_cat":"cs.CL","authors_text":"Furu Wei, Lei Cui, Qihao Zhao, Qinzheng Sun, Qiufeng Yin, Scarlett Li, Shaoguang Mao, Tengchao Lv, Xin Zhang, Yangyu Huang, Ying Xin","submitted_at":"2024-12-19T18:58:04Z","abstract_excerpt":"Multiple-choice question (MCQ) datasets like Massive Multitask Language Understanding (MMLU) are widely used to evaluate the commonsense, understanding, and problem-solving abilities of large language models (LLMs). However, the open-source nature of these benchmarks and the broad sources of training data for LLMs have inevitably led to benchmark contamination, resulting in unreliable evaluation results. To alleviate this issue, we propose a contamination-free and more challenging MCQ benchmark called MMLU-CF. This benchmark reassesses LLMs' understanding of world knowledge by averting both un"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.15194","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-12-19T18:58:04Z","cross_cats_sorted":["cs.AI","cs.LG","cs.PF"],"title_canon_sha256":"253427a712c66424579a51b7c0de2137b6eb8d06ce0259d84a5b3d5f74e8c2ba","abstract_canon_sha256":"c41233d72282e5e7bf8289eebb291c380b2639644c12fc66bee29537518e4293"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:27:52.223460Z","signature_b64":"Rp3gocnE2K/YuDeDh6loVndfUEDNLY9olzxEZWsPt498qC4gC4yvmKQKEqKDn8jYLOHPmcEtP3yBd3x352lSBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"63e8a15cfd206aa2b729489ec977be667a1222e07a30ad7ccc7b4577dd602e6d","last_reissued_at":"2026-07-05T11:27:52.222946Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:27:52.222946Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MMLU-CF: A Contamination-free Multi-task Language Understanding Benchmark","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.PF"],"primary_cat":"cs.CL","authors_text":"Furu Wei, Lei Cui, Qihao Zhao, Qinzheng Sun, Qiufeng Yin, Scarlett Li, Shaoguang Mao, Tengchao Lv, Xin Zhang, Yangyu Huang, Ying Xin","submitted_at":"2024-12-19T18:58:04Z","abstract_excerpt":"Multiple-choice question (MCQ) datasets like Massive Multitask Language Understanding (MMLU) are widely used to evaluate the commonsense, understanding, and problem-solving abilities of large language models (LLMs). However, the open-source nature of these benchmarks and the broad sources of training data for LLMs have inevitably led to benchmark contamination, resulting in unreliable evaluation results. To alleviate this issue, we propose a contamination-free and more challenging MCQ benchmark called MMLU-CF. This benchmark reassesses LLMs' understanding of world knowledge by averting both un"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.15194","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.15194/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.15194","created_at":"2026-07-05T11:27:52.223013+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.15194v1","created_at":"2026-07-05T11:27:52.223013+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.15194","created_at":"2026-07-05T11:27:52.223013+00:00"},{"alias_kind":"pith_short_12","alias_value":"MPUKCXH5EBVK","created_at":"2026-07-05T11:27:52.223013+00:00"},{"alias_kind":"pith_short_16","alias_value":"MPUKCXH5EBVKFNZJ","created_at":"2026-07-05T11:27:52.223013+00:00"},{"alias_kind":"pith_short_8","alias_value":"MPUKCXH5","created_at":"2026-07-05T11:27:52.223013+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26396","citing_title":"At the Edge of Understanding: Sparse Autoencoders Trace The Limits of Transformer Generalization","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21543","citing_title":"Provable Joint Decontamination for Benchmarking Multiple Large Language Models","ref_index":180,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04942","citing_title":"TDA-RC: Task-Driven Alignment for Knowledge-Based Reasoning Chains in Large Language Models","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15972","citing_title":"Weak-Link Optimization for Multi-Agent Reasoning and Collaboration","ref_index":55,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MPUKCXH5EBVKFNZJJCPMS556MZ","json":"https://pith.science/pith/MPUKCXH5EBVKFNZJJCPMS556MZ.json","graph_json":"https://pith.science/api/pith-number/MPUKCXH5EBVKFNZJJCPMS556MZ/graph.json","events_json":"https://pith.science/api/pith-number/MPUKCXH5EBVKFNZJJCPMS556MZ/events.json","paper":"https://pith.science/paper/MPUKCXH5"},"agent_actions":{"view_html":"https://pith.science/pith/MPUKCXH5EBVKFNZJJCPMS556MZ","download_json":"https://pith.science/pith/MPUKCXH5EBVKFNZJJCPMS556MZ.json","view_paper":"https://pith.science/paper/MPUKCXH5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.15194&json=true","fetch_graph":"https://pith.science/api/pith-number/MPUKCXH5EBVKFNZJJCPMS556MZ/graph.json","fetch_events":"https://pith.science/api/pith-number/MPUKCXH5EBVKFNZJJCPMS556MZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MPUKCXH5EBVKFNZJJCPMS556MZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MPUKCXH5EBVKFNZJJCPMS556MZ/action/storage_attestation","attest_author":"https://pith.science/pith/MPUKCXH5EBVKFNZJJCPMS556MZ/action/author_attestation","sign_citation":"https://pith.science/pith/MPUKCXH5EBVKFNZJJCPMS556MZ/action/citation_signature","submit_replication":"https://pith.science/pith/MPUKCXH5EBVKFNZJJCPMS556MZ/action/replication_record"}},"created_at":"2026-07-05T11:27:52.223013+00:00","updated_at":"2026-07-05T11:27:52.223013+00:00"}