{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:6UOJ7OL354FGOSOOL4LDV33ISE","short_pith_number":"pith:6UOJ7OL3","schema_version":"1.0","canonical_sha256":"f51c9fb97bef0a6749ce5f163aef689125ce3d2eb550b88e850d79fbe9856188","source":{"kind":"arxiv","id":"2410.01999","version":4},"attestation_state":"computed","paper":{"title":"CodeMMLU: A Multi-Task Benchmark for Assessing Code Understanding & Reasoning Capabilities of CodeLLMs","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Dung Nguyen Manh, Nam Le Hai, Nam V. Nguyen, Nghi D. Q. Bui, Quang Pham, Thang Phan Chau, Thong T. Doan","submitted_at":"2024-10-02T20:04:02Z","abstract_excerpt":"Recent advances in Code Large Language Models (CodeLLMs) have primarily focused on open-ended code generation, often overlooking the crucial aspect of code understanding and reasoning. To bridge this gap, we introduce CodeMMLU, a comprehensive multiple-choice benchmark designed to evaluate the depth of software and code comprehension in LLMs. CodeMMLU includes nearly 20,000 questions spanning diverse domains, including code analysis, defect detection, and software engineering principles across multiple programming languages. Unlike traditional benchmarks that emphasize code generation, CodeMML"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.01999","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"cs.SE","submitted_at":"2024-10-02T20:04:02Z","cross_cats_sorted":[],"title_canon_sha256":"57592ccf9f1852bacd188b5d95a8351042d32cf463b3f5f6b444612516e33cda","abstract_canon_sha256":"420495cf40720bc9749884f2255eba0dd09b0aeb67905e1df339ad3415fd59cb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:46:38.656838Z","signature_b64":"mwSOXfZkwiX+LjLLeerHprF7WiMITc0bdiCRyGNZz5MDGcIqXIlyCqTU4O27vDAMaG7TsZQD024mDTZaCiobDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f51c9fb97bef0a6749ce5f163aef689125ce3d2eb550b88e850d79fbe9856188","last_reissued_at":"2026-07-05T10:46:38.656378Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:46:38.656378Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CodeMMLU: A Multi-Task Benchmark for Assessing Code Understanding & Reasoning Capabilities of CodeLLMs","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Dung Nguyen Manh, Nam Le Hai, Nam V. Nguyen, Nghi D. Q. Bui, Quang Pham, Thang Phan Chau, Thong T. Doan","submitted_at":"2024-10-02T20:04:02Z","abstract_excerpt":"Recent advances in Code Large Language Models (CodeLLMs) have primarily focused on open-ended code generation, often overlooking the crucial aspect of code understanding and reasoning. To bridge this gap, we introduce CodeMMLU, a comprehensive multiple-choice benchmark designed to evaluate the depth of software and code comprehension in LLMs. CodeMMLU includes nearly 20,000 questions spanning diverse domains, including code analysis, defect detection, and software engineering principles across multiple programming languages. Unlike traditional benchmarks that emphasize code generation, CodeMML"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.01999","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.01999/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.01999","created_at":"2026-07-05T10:46:38.656440+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.01999v4","created_at":"2026-07-05T10:46:38.656440+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.01999","created_at":"2026-07-05T10:46:38.656440+00:00"},{"alias_kind":"pith_short_12","alias_value":"6UOJ7OL354FG","created_at":"2026-07-05T10:46:38.656440+00:00"},{"alias_kind":"pith_short_16","alias_value":"6UOJ7OL354FGOSOO","created_at":"2026-07-05T10:46:38.656440+00:00"},{"alias_kind":"pith_short_8","alias_value":"6UOJ7OL3","created_at":"2026-07-05T10:46:38.656440+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.20857","citing_title":"DiagramBank: A Large-scale Dataset of Diagram Design Exemplars with Paper Metadata for Retrieval-Augmented Generation","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6UOJ7OL354FGOSOOL4LDV33ISE","json":"https://pith.science/pith/6UOJ7OL354FGOSOOL4LDV33ISE.json","graph_json":"https://pith.science/api/pith-number/6UOJ7OL354FGOSOOL4LDV33ISE/graph.json","events_json":"https://pith.science/api/pith-number/6UOJ7OL354FGOSOOL4LDV33ISE/events.json","paper":"https://pith.science/paper/6UOJ7OL3"},"agent_actions":{"view_html":"https://pith.science/pith/6UOJ7OL354FGOSOOL4LDV33ISE","download_json":"https://pith.science/pith/6UOJ7OL354FGOSOOL4LDV33ISE.json","view_paper":"https://pith.science/paper/6UOJ7OL3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.01999&json=true","fetch_graph":"https://pith.science/api/pith-number/6UOJ7OL354FGOSOOL4LDV33ISE/graph.json","fetch_events":"https://pith.science/api/pith-number/6UOJ7OL354FGOSOOL4LDV33ISE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6UOJ7OL354FGOSOOL4LDV33ISE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6UOJ7OL354FGOSOOL4LDV33ISE/action/storage_attestation","attest_author":"https://pith.science/pith/6UOJ7OL354FGOSOOL4LDV33ISE/action/author_attestation","sign_citation":"https://pith.science/pith/6UOJ7OL354FGOSOOL4LDV33ISE/action/citation_signature","submit_replication":"https://pith.science/pith/6UOJ7OL354FGOSOOL4LDV33ISE/action/replication_record"}},"created_at":"2026-07-05T10:46:38.656440+00:00","updated_at":"2026-07-05T10:46:38.656440+00:00"}