{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:APGIE3RXEUCS7VY2YXJKWJOLNU","short_pith_number":"pith:APGIE3RX","schema_version":"1.0","canonical_sha256":"03cc826e3725052fd71ac5d2ab25cb6d0a3b5ab45743fe97e1540a485046204b","source":{"kind":"arxiv","id":"2406.04531","version":2},"attestation_state":"computed","paper":{"title":"TESTEVAL: Benchmarking Large Language Models for Test Case Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"An Ran Chen, Chenyuan Yang, Da Song, Lei Ma, Lingming Zhang, Wenhan Wang, Yuheng Huang, Zhaoyang Chu, Zhijie Wang","submitted_at":"2024-06-06T22:07:50Z","abstract_excerpt":"Testing plays a crucial role in the software development cycle, enabling the detection of bugs, vulnerabilities, and other undesirable behaviors. To perform software testing, testers need to write code snippets that execute the program under test. Recently, researchers have recognized the potential of large language models (LLMs) in software testing. However, there remains a lack of fair comparisons between different LLMs in terms of test case generation capabilities.\n  In this paper, we propose TESTEVAL, a novel benchmark for test case generation with LLMs. We collect 210 Python programs from"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.04531","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SE","submitted_at":"2024-06-06T22:07:50Z","cross_cats_sorted":[],"title_canon_sha256":"36a039ee539c0e23d3584e30cc3b6ecb24963c6b93f5a1b2374ef55d4f89a89c","abstract_canon_sha256":"1df7ad2d4f077b0f7425e41356cabf766d6a5e23d56c17825399df13ac3bb5c7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:08:17.733960Z","signature_b64":"O3PbRrPrA6ckl4b/6edYs7V2Vow5qo9XgcI7LbJ+6TPj0Xu7p0g8THhE6yMabqsb3nht+H95dQY5Sg1rYtdjBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"03cc826e3725052fd71ac5d2ab25cb6d0a3b5ab45743fe97e1540a485046204b","last_reissued_at":"2026-07-05T10:08:17.733382Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:08:17.733382Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TESTEVAL: Benchmarking Large Language Models for Test Case Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"An Ran Chen, Chenyuan Yang, Da Song, Lei Ma, Lingming Zhang, Wenhan Wang, Yuheng Huang, Zhaoyang Chu, Zhijie Wang","submitted_at":"2024-06-06T22:07:50Z","abstract_excerpt":"Testing plays a crucial role in the software development cycle, enabling the detection of bugs, vulnerabilities, and other undesirable behaviors. To perform software testing, testers need to write code snippets that execute the program under test. Recently, researchers have recognized the potential of large language models (LLMs) in software testing. However, there remains a lack of fair comparisons between different LLMs in terms of test case generation capabilities.\n  In this paper, we propose TESTEVAL, a novel benchmark for test case generation with LLMs. We collect 210 Python programs from"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.04531","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.04531/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.04531","created_at":"2026-07-05T10:08:17.733471+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.04531v2","created_at":"2026-07-05T10:08:17.733471+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.04531","created_at":"2026-07-05T10:08:17.733471+00:00"},{"alias_kind":"pith_short_12","alias_value":"APGIE3RXEUCS","created_at":"2026-07-05T10:08:17.733471+00:00"},{"alias_kind":"pith_short_16","alias_value":"APGIE3RXEUCS7VY2","created_at":"2026-07-05T10:08:17.733471+00:00"},{"alias_kind":"pith_short_8","alias_value":"APGIE3RX","created_at":"2026-07-05T10:08:17.733471+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2502.06556","citing_title":"MultiFileTest: A Multi-File-Level LLM Unit Test Generation Benchmark and Impact of Error Fixing Mechanisms","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2512.18470","citing_title":"SWE-EVO: Benchmarking Coding Agents in Long-Horizon Software Evolution Scenarios","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13139","citing_title":"SWE-Cycle: Benchmarking Code Agents across the Complete Issue Resolution Cycle","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13559","citing_title":"WebMAC: A Multi-Agent Collaborative Framework for Scenario Testing of Web Systems","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15270","citing_title":"Enhancing Large Language Models with Retrieval Augmented Generation for Software Testing and Inspection Automation","ref_index":11,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/APGIE3RXEUCS7VY2YXJKWJOLNU","json":"https://pith.science/pith/APGIE3RXEUCS7VY2YXJKWJOLNU.json","graph_json":"https://pith.science/api/pith-number/APGIE3RXEUCS7VY2YXJKWJOLNU/graph.json","events_json":"https://pith.science/api/pith-number/APGIE3RXEUCS7VY2YXJKWJOLNU/events.json","paper":"https://pith.science/paper/APGIE3RX"},"agent_actions":{"view_html":"https://pith.science/pith/APGIE3RXEUCS7VY2YXJKWJOLNU","download_json":"https://pith.science/pith/APGIE3RXEUCS7VY2YXJKWJOLNU.json","view_paper":"https://pith.science/paper/APGIE3RX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.04531&json=true","fetch_graph":"https://pith.science/api/pith-number/APGIE3RXEUCS7VY2YXJKWJOLNU/graph.json","fetch_events":"https://pith.science/api/pith-number/APGIE3RXEUCS7VY2YXJKWJOLNU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/APGIE3RXEUCS7VY2YXJKWJOLNU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/APGIE3RXEUCS7VY2YXJKWJOLNU/action/storage_attestation","attest_author":"https://pith.science/pith/APGIE3RXEUCS7VY2YXJKWJOLNU/action/author_attestation","sign_citation":"https://pith.science/pith/APGIE3RXEUCS7VY2YXJKWJOLNU/action/citation_signature","submit_replication":"https://pith.science/pith/APGIE3RXEUCS7VY2YXJKWJOLNU/action/replication_record"}},"created_at":"2026-07-05T10:08:17.733471+00:00","updated_at":"2026-07-05T10:08:17.733471+00:00"}