{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:IO4ZESJ54XR7XFDTHQAB2FGLLU","short_pith_number":"pith:IO4ZESJ5","schema_version":"1.0","canonical_sha256":"43b992493de5e3fb94733c001d14cb5d05894272107236254bfbbf1379894ce3","source":{"kind":"arxiv","id":"2402.07844","version":4},"attestation_state":"computed","paper":{"title":"Mercury: A Code Efficiency Benchmark for Code Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.SE","authors_text":"Anh Tuan Luu, Bin Ji, Mingzhe Du, Qian Liu, See-kiong Ng","submitted_at":"2024-02-12T17:53:22Z","abstract_excerpt":"Amidst the recent strides in evaluating Large Language Models for Code (Code LLMs), existing benchmarks have mainly focused on the functional correctness of generated code, neglecting the importance of their computational efficiency. To fill the gap, we present Mercury, the first code efficiency benchmark for Code LLMs. It comprises 1,889 Python tasks, each accompanied by adequate solutions that serve as real-world efficiency baselines, enabling a comprehensive analysis of the runtime distribution. Based on the distribution, we introduce a new metric Beyond, which computes a runtime-percentile"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.07844","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SE","submitted_at":"2024-02-12T17:53:22Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"f0750fef0643fd3dada7e794c5cb5272fdcaa7f93b744225465a529a02bf577c","abstract_canon_sha256":"b2ea03c31f057c670d4b3c3f7265ea05fc2bebd6ef557da8d56e4f222e49396f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:30:08.521185Z","signature_b64":"CYBpL1FBj71mlQLdpY5P0isiLowA8Xhiiv5veTWSwZ27ZVwPiOV1aviZIVCvNggP605HwMz9OiP/khsmHdWXDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"43b992493de5e3fb94733c001d14cb5d05894272107236254bfbbf1379894ce3","last_reissued_at":"2026-07-05T08:30:08.520711Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:30:08.520711Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Mercury: A Code Efficiency Benchmark for Code Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.SE","authors_text":"Anh Tuan Luu, Bin Ji, Mingzhe Du, Qian Liu, See-kiong Ng","submitted_at":"2024-02-12T17:53:22Z","abstract_excerpt":"Amidst the recent strides in evaluating Large Language Models for Code (Code LLMs), existing benchmarks have mainly focused on the functional correctness of generated code, neglecting the importance of their computational efficiency. To fill the gap, we present Mercury, the first code efficiency benchmark for Code LLMs. It comprises 1,889 Python tasks, each accompanied by adequate solutions that serve as real-world efficiency baselines, enabling a comprehensive analysis of the runtime distribution. Based on the distribution, we introduce a new metric Beyond, which computes a runtime-percentile"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.07844","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.07844/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.07844","created_at":"2026-07-05T08:30:08.520765+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.07844v4","created_at":"2026-07-05T08:30:08.520765+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.07844","created_at":"2026-07-05T08:30:08.520765+00:00"},{"alias_kind":"pith_short_12","alias_value":"IO4ZESJ54XR7","created_at":"2026-07-05T08:30:08.520765+00:00"},{"alias_kind":"pith_short_16","alias_value":"IO4ZESJ54XR7XFDT","created_at":"2026-07-05T08:30:08.520765+00:00"},{"alias_kind":"pith_short_8","alias_value":"IO4ZESJ5","created_at":"2026-07-05T08:30:08.520765+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.05249","citing_title":"SWE-InfraBench: Evaluating Language Models on Cloud Infrastructure Code","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2411.10656","citing_title":"Precision or Peril: A PoC of Python Code Quality from Quantized Large Language Models","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2512.14018","citing_title":"PerfCoder: Large Language Models for Interpretable Code Performance Optimization","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03144","citing_title":"InCoder-32B-Thinking: Industrial Code World Model for Thinking","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05267","citing_title":"Bridging Generation and Training: A Systematic Review of Quality Issues in LLMs for Code","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IO4ZESJ54XR7XFDTHQAB2FGLLU","json":"https://pith.science/pith/IO4ZESJ54XR7XFDTHQAB2FGLLU.json","graph_json":"https://pith.science/api/pith-number/IO4ZESJ54XR7XFDTHQAB2FGLLU/graph.json","events_json":"https://pith.science/api/pith-number/IO4ZESJ54XR7XFDTHQAB2FGLLU/events.json","paper":"https://pith.science/paper/IO4ZESJ5"},"agent_actions":{"view_html":"https://pith.science/pith/IO4ZESJ54XR7XFDTHQAB2FGLLU","download_json":"https://pith.science/pith/IO4ZESJ54XR7XFDTHQAB2FGLLU.json","view_paper":"https://pith.science/paper/IO4ZESJ5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.07844&json=true","fetch_graph":"https://pith.science/api/pith-number/IO4ZESJ54XR7XFDTHQAB2FGLLU/graph.json","fetch_events":"https://pith.science/api/pith-number/IO4ZESJ54XR7XFDTHQAB2FGLLU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IO4ZESJ54XR7XFDTHQAB2FGLLU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IO4ZESJ54XR7XFDTHQAB2FGLLU/action/storage_attestation","attest_author":"https://pith.science/pith/IO4ZESJ54XR7XFDTHQAB2FGLLU/action/author_attestation","sign_citation":"https://pith.science/pith/IO4ZESJ54XR7XFDTHQAB2FGLLU/action/citation_signature","submit_replication":"https://pith.science/pith/IO4ZESJ54XR7XFDTHQAB2FGLLU/action/replication_record"}},"created_at":"2026-07-05T08:30:08.520765+00:00","updated_at":"2026-07-05T08:30:08.520765+00:00"}