{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:IKCRTACL7PGNFPLPVVANJA4TZQ","short_pith_number":"pith:IKCRTACL","schema_version":"1.0","canonical_sha256":"428519804bfbccd2bd6fad40d48393cc10c5430327dc1e23018306a5bdb12dc6","source":{"kind":"arxiv","id":"2407.11470","version":2},"attestation_state":"computed","paper":{"title":"Beyond Correctness: Benchmarking Multi-dimensional Code Generation for Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.SE","authors_text":"Boxi Cao, Hongyu Lin, Jiasheng Zheng, Le Sun, Ruotong Pan, Xianpei Han, Yaojie Lu, Zhengzhao Ma","submitted_at":"2024-07-16T08:08:48Z","abstract_excerpt":"In recent years, researchers have proposed numerous benchmarks to evaluate the impressive coding capabilities of large language models (LLMs). However, current benchmarks primarily assess the accuracy of LLM-generated code, while neglecting other critical dimensions that also significantly impact code quality in real-world development. Moreover, relying exclusively on correctness as the guiding metric renders LLMs susceptible to data contamination. Therefore, this paper proposes the RACE benchmark, which comprehensively evaluates the quality of code generated by LLMs across 4 dimensions: Reada"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.11470","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SE","submitted_at":"2024-07-16T08:08:48Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"4b1f339e194f26cba3dc08ea440e8b65beec69800f6bee669e2e57874c470379","abstract_canon_sha256":"d96949756b13d9f3babdc14e2c39bd49df9f437a968752f5fc92b4e29a0ba1c6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:17:45.999503Z","signature_b64":"yog027GxpQ/nj7EKNKgfA8NES8r1M4bT1ZkmlEBZ5tv1UmCe8UeEHlofLQ92GYGsSnmUmmmPGev/vCJ99bIkCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"428519804bfbccd2bd6fad40d48393cc10c5430327dc1e23018306a5bdb12dc6","last_reissued_at":"2026-07-05T09:17:45.999015Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:17:45.999015Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Beyond Correctness: Benchmarking Multi-dimensional Code Generation for Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.SE","authors_text":"Boxi Cao, Hongyu Lin, Jiasheng Zheng, Le Sun, Ruotong Pan, Xianpei Han, Yaojie Lu, Zhengzhao Ma","submitted_at":"2024-07-16T08:08:48Z","abstract_excerpt":"In recent years, researchers have proposed numerous benchmarks to evaluate the impressive coding capabilities of large language models (LLMs). However, current benchmarks primarily assess the accuracy of LLM-generated code, while neglecting other critical dimensions that also significantly impact code quality in real-world development. Moreover, relying exclusively on correctness as the guiding metric renders LLMs susceptible to data contamination. Therefore, this paper proposes the RACE benchmark, which comprehensively evaluates the quality of code generated by LLMs across 4 dimensions: Reada"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.11470","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.11470/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.11470","created_at":"2026-07-05T09:17:45.999070+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.11470v2","created_at":"2026-07-05T09:17:45.999070+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.11470","created_at":"2026-07-05T09:17:45.999070+00:00"},{"alias_kind":"pith_short_12","alias_value":"IKCRTACL7PGN","created_at":"2026-07-05T09:17:45.999070+00:00"},{"alias_kind":"pith_short_16","alias_value":"IKCRTACL7PGNFPLP","created_at":"2026-07-05T09:17:45.999070+00:00"},{"alias_kind":"pith_short_8","alias_value":"IKCRTACL","created_at":"2026-07-05T09:17:45.999070+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08009","citing_title":"From Execution to Education: A Bloom-Aligned Framework for Measuring Educational Control in LLMs","ref_index":206,"is_internal_anchor":true},{"citing_arxiv_id":"2606.08588","citing_title":"LLM vs. Human Unit Tests: Fault Detection on Real Python Bugs","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25296","citing_title":"Subjective Code Preferences in Experts and Large Language Models","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2507.11687","citing_title":"MetaLint: Easy-to-Hard Generalization for Code Linting","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2510.15494","citing_title":"Do AI Models Dream of Faster Code? An Empirical Study on LLM-Proposed Performance Improvements in Real-World Software","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13776","citing_title":"\"Like Taking the Path of Least Resistance\": Exploring the Impact of LLM Interaction on the Creative Process of Programming","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05267","citing_title":"Bridging Generation and Training: A Systematic Review of Quality Issues in LLMs for Code","ref_index":154,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IKCRTACL7PGNFPLPVVANJA4TZQ","json":"https://pith.science/pith/IKCRTACL7PGNFPLPVVANJA4TZQ.json","graph_json":"https://pith.science/api/pith-number/IKCRTACL7PGNFPLPVVANJA4TZQ/graph.json","events_json":"https://pith.science/api/pith-number/IKCRTACL7PGNFPLPVVANJA4TZQ/events.json","paper":"https://pith.science/paper/IKCRTACL"},"agent_actions":{"view_html":"https://pith.science/pith/IKCRTACL7PGNFPLPVVANJA4TZQ","download_json":"https://pith.science/pith/IKCRTACL7PGNFPLPVVANJA4TZQ.json","view_paper":"https://pith.science/paper/IKCRTACL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.11470&json=true","fetch_graph":"https://pith.science/api/pith-number/IKCRTACL7PGNFPLPVVANJA4TZQ/graph.json","fetch_events":"https://pith.science/api/pith-number/IKCRTACL7PGNFPLPVVANJA4TZQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IKCRTACL7PGNFPLPVVANJA4TZQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IKCRTACL7PGNFPLPVVANJA4TZQ/action/storage_attestation","attest_author":"https://pith.science/pith/IKCRTACL7PGNFPLPVVANJA4TZQ/action/author_attestation","sign_citation":"https://pith.science/pith/IKCRTACL7PGNFPLPVVANJA4TZQ/action/citation_signature","submit_replication":"https://pith.science/pith/IKCRTACL7PGNFPLPVVANJA4TZQ/action/replication_record"}},"created_at":"2026-07-05T09:17:45.999070+00:00","updated_at":"2026-07-05T09:17:45.999070+00:00"}