{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:2UF75TJCFO5HTC6DXTAIU24XO6","short_pith_number":"pith:2UF75TJC","schema_version":"1.0","canonical_sha256":"d50bfecd222bba798bc3bcc08a6b9777aeae9978d2c0955d566ded2bceec5486","source":{"kind":"arxiv","id":"2308.01861","version":2},"attestation_state":"computed","paper":{"title":"ClassEval: A Manually-Crafted Benchmark for Evaluating LLMs on Class-level Code Generation","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Chaofeng Sha, Hanlin Wang, Jiayi Feng, Junwei Liu, Kaixin Wang, Mingwei Liu, Xin Peng, Xueying Du, Yiling Lou, Yixuan Chen","submitted_at":"2023-08-03T16:31:02Z","abstract_excerpt":"In this work, we make the first attempt to evaluate LLMs in a more challenging code generation scenario, i.e. class-level code generation. We first manually construct the first class-level code generation benchmark ClassEval of 100 class-level Python code generation tasks with approximately 500 person-hours. Based on it, we then perform the first study of 11 state-of-the-art LLMs on class-level code generation. Based on our results, we have the following main findings. First, we find that all existing LLMs show much worse performance on class-level code generation compared to on standalone met"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.01861","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2023-08-03T16:31:02Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"f228f207f677b3c3cd103c39615b9994aa3ebdb450c9619585eb35b85b42d829","abstract_canon_sha256":"0ff4fbcd91c3bb6d2f3d972c6fa975b0f9bab15a0eb77b64575479c0b9a64669"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:40:54.008469Z","signature_b64":"1MwD5HhGq7DuNeKvgl2AmY/F+JB0Cg+STqoUUHiZaaAd8lL8PX1BsZ8EkudFS1TMK2bQyPJy1BXiNWaUwYFoCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d50bfecd222bba798bc3bcc08a6b9777aeae9978d2c0955d566ded2bceec5486","last_reissued_at":"2026-07-05T06:40:54.007884Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:40:54.007884Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ClassEval: A Manually-Crafted Benchmark for Evaluating LLMs on Class-level Code Generation","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Chaofeng Sha, Hanlin Wang, Jiayi Feng, Junwei Liu, Kaixin Wang, Mingwei Liu, Xin Peng, Xueying Du, Yiling Lou, Yixuan Chen","submitted_at":"2023-08-03T16:31:02Z","abstract_excerpt":"In this work, we make the first attempt to evaluate LLMs in a more challenging code generation scenario, i.e. class-level code generation. We first manually construct the first class-level code generation benchmark ClassEval of 100 class-level Python code generation tasks with approximately 500 person-hours. Based on it, we then perform the first study of 11 state-of-the-art LLMs on class-level code generation. Based on our results, we have the following main findings. First, we find that all existing LLMs show much worse performance on class-level code generation compared to on standalone met"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.01861","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.01861/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.01861","created_at":"2026-07-05T06:40:54.007947+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.01861v2","created_at":"2026-07-05T06:40:54.007947+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.01861","created_at":"2026-07-05T06:40:54.007947+00:00"},{"alias_kind":"pith_short_12","alias_value":"2UF75TJCFO5H","created_at":"2026-07-05T06:40:54.007947+00:00"},{"alias_kind":"pith_short_16","alias_value":"2UF75TJCFO5HTC6D","created_at":"2026-07-05T06:40:54.007947+00:00"},{"alias_kind":"pith_short_8","alias_value":"2UF75TJC","created_at":"2026-07-05T06:40:54.007947+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":23,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07619","citing_title":"Rethinking Code Performance Benchmarks for LLMs","ref_index":147,"is_internal_anchor":true},{"citing_arxiv_id":"2606.20835","citing_title":"PromptMark: A Prompt-Guided Iterative-Feedback Framework for Source Code Watermarking","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08676","citing_title":"Lost in the Flow with Code Talkers: Unveiling the Instruction-Tuning Tax of Large Language Models in Code Tasks","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08840","citing_title":"Beyond Pass Rate: A Multilingual, Execution-Grounded Evaluation of Open Code LLMs","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28998","citing_title":"Reward-Free Code Alignment from Pretrained or Fine-Tuned LLM: Unpacking the Trade-offs for Code Generation","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2508.15503","citing_title":"Guidelines for Empirical Studies in Software Engineering involving Large Language Models","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23262","citing_title":"Design and Report Benchmarks for Knowledge Work","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2402.09664","citing_title":"CodeMind: Evaluating Large Language Models for Code Reasoning","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2506.08980","citing_title":"AdaDec: A Uncertainty-Guided Lookahead Decoding Framework for LLM-Based Code Generation","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2508.15503","citing_title":"Guidelines for Empirical Studies in Software Engineering involving Large Language Models","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2512.00380","citing_title":"Knowledge-Graph-Driven Data Synthesis for Low-Resource Software Development: A HarmonyOS Case Study","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2512.14917","citing_title":"Evaluating Code Reasoning Abilities of Large Language Models Under Real-World Settings","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16322","citing_title":"Steerable Instruction Following Coding Data Synthesis with Actor-Parametric Schema Co-Evolution","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2502.18449","citing_title":"SWE-RL: Advancing LLM Reasoning via Reinforcement Learning on Open Software Evolution","ref_index":166,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26923","citing_title":"ClassEval-Pro: A Cross-Domain Benchmark for Class-Level Code Generation","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22659","citing_title":"RealBench: A Repo-Level Code Generation Benchmark Aligned with Real-World Software Development Practices","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04637","citing_title":"SWE-WebDevBench: Evaluating Coding Agent Application Platforms as Virtual Software Agencies","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10520","citing_title":"ReFEree: Reference-Free and Fine-Grained Method for Evaluating Factual Consistency in Real-World Code Summarization","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10481","citing_title":"PatchRecall: Patch-Driven Retrieval for Automated Program Repair","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06683","citing_title":"Benchmarking Requirement-to-Architecture Generation with Hybrid Evaluation","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07864","citing_title":"ZeroCoder: Can LLMs Improve Code Generation Without Ground-Truth Supervision?","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12881","citing_title":"Evaluating LLMs Code Reasoning Under Real-World Context","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2504.15564","citing_title":"OpenClassGen: A Large-Scale Corpus of Real-World Python Classes for LLM Research","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2UF75TJCFO5HTC6DXTAIU24XO6","json":"https://pith.science/pith/2UF75TJCFO5HTC6DXTAIU24XO6.json","graph_json":"https://pith.science/api/pith-number/2UF75TJCFO5HTC6DXTAIU24XO6/graph.json","events_json":"https://pith.science/api/pith-number/2UF75TJCFO5HTC6DXTAIU24XO6/events.json","paper":"https://pith.science/paper/2UF75TJC"},"agent_actions":{"view_html":"https://pith.science/pith/2UF75TJCFO5HTC6DXTAIU24XO6","download_json":"https://pith.science/pith/2UF75TJCFO5HTC6DXTAIU24XO6.json","view_paper":"https://pith.science/paper/2UF75TJC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.01861&json=true","fetch_graph":"https://pith.science/api/pith-number/2UF75TJCFO5HTC6DXTAIU24XO6/graph.json","fetch_events":"https://pith.science/api/pith-number/2UF75TJCFO5HTC6DXTAIU24XO6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2UF75TJCFO5HTC6DXTAIU24XO6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2UF75TJCFO5HTC6DXTAIU24XO6/action/storage_attestation","attest_author":"https://pith.science/pith/2UF75TJCFO5HTC6DXTAIU24XO6/action/author_attestation","sign_citation":"https://pith.science/pith/2UF75TJCFO5HTC6DXTAIU24XO6/action/citation_signature","submit_replication":"https://pith.science/pith/2UF75TJCFO5HTC6DXTAIU24XO6/action/replication_record"}},"created_at":"2026-07-05T06:40:54.007947+00:00","updated_at":"2026-07-05T06:40:54.007947+00:00"}