{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:KGGC6CAI7QYYUBZA6XWZIIFR2J","short_pith_number":"pith:KGGC6CAI","schema_version":"1.0","canonical_sha256":"518c2f0808fc318a0720f5ed9420b1d243c8198315f4547175daa1b35a4ac79c","source":{"kind":"arxiv","id":"2406.19470","version":2},"attestation_state":"computed","paper":{"title":"Changing Answer Order Can Decrease MMLU Accuracy","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Adina Williams, Candace Ross, David Pantoja, Megan Ung, Vipul Gupta","submitted_at":"2024-06-27T18:21:32Z","abstract_excerpt":"As large language models (LLMs) have grown in prevalence, particular benchmarks have become essential for the evaluation of these models and for understanding model capabilities. Most commonly, we use test accuracy averaged across multiple subtasks in order to rank models on leaderboards, to determine which model is best for our purposes. In this paper, we investigate the robustness of the accuracy measurement on a widely used multiple choice question answering dataset, MMLU. When shuffling the answer label contents, we find that all explored models decrease in accuracy on MMLU, but not every "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.19470","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-27T18:21:32Z","cross_cats_sorted":[],"title_canon_sha256":"5e5fc3bacdded614758f3267dedd099b6cab704301ee440fb42c2f87f5b8be0a","abstract_canon_sha256":"082585537ee56c6910c94c4463fbf79d72ed991ac3b0f92d142da82456c972b2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:33:34.161702Z","signature_b64":"MRVQKyar4eextE7sqlnN2U6zjo2AADgNT6hEhrNbr8ck55/Ef2AoczrZ8jMLDMMxjitiAanRzyeZ438nUpw8Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"518c2f0808fc318a0720f5ed9420b1d243c8198315f4547175daa1b35a4ac79c","last_reissued_at":"2026-07-05T09:33:34.161203Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:33:34.161203Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Changing Answer Order Can Decrease MMLU Accuracy","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Adina Williams, Candace Ross, David Pantoja, Megan Ung, Vipul Gupta","submitted_at":"2024-06-27T18:21:32Z","abstract_excerpt":"As large language models (LLMs) have grown in prevalence, particular benchmarks have become essential for the evaluation of these models and for understanding model capabilities. Most commonly, we use test accuracy averaged across multiple subtasks in order to rank models on leaderboards, to determine which model is best for our purposes. In this paper, we investigate the robustness of the accuracy measurement on a widely used multiple choice question answering dataset, MMLU. When shuffling the answer label contents, we find that all explored models decrease in accuracy on MMLU, but not every "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.19470","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.19470/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.19470","created_at":"2026-07-05T09:33:34.161266+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.19470v2","created_at":"2026-07-05T09:33:34.161266+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.19470","created_at":"2026-07-05T09:33:34.161266+00:00"},{"alias_kind":"pith_short_12","alias_value":"KGGC6CAI7QYY","created_at":"2026-07-05T09:33:34.161266+00:00"},{"alias_kind":"pith_short_16","alias_value":"KGGC6CAI7QYYUBZA","created_at":"2026-07-05T09:33:34.161266+00:00"},{"alias_kind":"pith_short_8","alias_value":"KGGC6CAI","created_at":"2026-07-05T09:33:34.161266+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.15607","citing_title":"Syntax Without Semantics: Teaching Large Language Models to Code in an Unseen Language","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2507.23009","citing_title":"Position: Stop Evaluating AI with Human Tests, Develop Principled, AI-specific Tests instead","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2603.09678","citing_title":"EsoLang-Bench: Evaluating Genuine Reasoning in Large Language Models via Esoteric Programming Languages","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07268","citing_title":"From 0-Order Selection to 2-Order Judgment: Combinatorial Hardening Exposes Compositional Failures in Frontier LLMs","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KGGC6CAI7QYYUBZA6XWZIIFR2J","json":"https://pith.science/pith/KGGC6CAI7QYYUBZA6XWZIIFR2J.json","graph_json":"https://pith.science/api/pith-number/KGGC6CAI7QYYUBZA6XWZIIFR2J/graph.json","events_json":"https://pith.science/api/pith-number/KGGC6CAI7QYYUBZA6XWZIIFR2J/events.json","paper":"https://pith.science/paper/KGGC6CAI"},"agent_actions":{"view_html":"https://pith.science/pith/KGGC6CAI7QYYUBZA6XWZIIFR2J","download_json":"https://pith.science/pith/KGGC6CAI7QYYUBZA6XWZIIFR2J.json","view_paper":"https://pith.science/paper/KGGC6CAI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.19470&json=true","fetch_graph":"https://pith.science/api/pith-number/KGGC6CAI7QYYUBZA6XWZIIFR2J/graph.json","fetch_events":"https://pith.science/api/pith-number/KGGC6CAI7QYYUBZA6XWZIIFR2J/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KGGC6CAI7QYYUBZA6XWZIIFR2J/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KGGC6CAI7QYYUBZA6XWZIIFR2J/action/storage_attestation","attest_author":"https://pith.science/pith/KGGC6CAI7QYYUBZA6XWZIIFR2J/action/author_attestation","sign_citation":"https://pith.science/pith/KGGC6CAI7QYYUBZA6XWZIIFR2J/action/citation_signature","submit_replication":"https://pith.science/pith/KGGC6CAI7QYYUBZA6XWZIIFR2J/action/replication_record"}},"created_at":"2026-07-05T09:33:34.161266+00:00","updated_at":"2026-07-05T09:33:34.161266+00:00"}