{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ED5MISNOIMUPTYO5K3PQYLJ5OW","short_pith_number":"pith:ED5MISNO","schema_version":"1.0","canonical_sha256":"20fac449ae4328f9e1dd56df0c2d3d75aeeb307ec523c4ad0ce3694c07eda901","source":{"kind":"arxiv","id":"2409.17561","version":1},"attestation_state":"computed","paper":{"title":"TestBench: Evaluating Class-Level Test Case Generation Capability of Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Chunrong Fang, Jianyi Zhou, Quanjun Zhang, Siqi Gu, Ye Shang, Zhenyu Chen","submitted_at":"2024-09-26T06:18:06Z","abstract_excerpt":"Software testing is a crucial phase in the software life cycle, helping identify potential risks and reduce maintenance costs. With the advancement of Large Language Models (LLMs), researchers have proposed an increasing number of LLM-based software testing techniques, particularly in the area of test case generation. Despite the growing interest, limited efforts have been made to thoroughly evaluate the actual capabilities of LLMs in this task.\n  In this paper, we introduce TestBench, a benchmark for class-level LLM-based test case generation. We construct a dataset of 108 Java programs from "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.17561","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SE","submitted_at":"2024-09-26T06:18:06Z","cross_cats_sorted":[],"title_canon_sha256":"16da0c9c3a06799a5326771b70cbf1293810ecd10321ca89d99982ea87428c38","abstract_canon_sha256":"6892c02137cdde76f0487d87dc005de653fc821df087fa68b4d76c2a4c010ac1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:12:11.905669Z","signature_b64":"wlUI1MitAU0n65IuYZW0z7MqCOkm4aSL17Fbi8+SF8V/nqGjPuj/fEz+ABjurzAmjGr4rBD5h1CtEg28tMk3Dw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"20fac449ae4328f9e1dd56df0c2d3d75aeeb307ec523c4ad0ce3694c07eda901","last_reissued_at":"2026-07-05T09:12:11.905195Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:12:11.905195Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TestBench: Evaluating Class-Level Test Case Generation Capability of Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Chunrong Fang, Jianyi Zhou, Quanjun Zhang, Siqi Gu, Ye Shang, Zhenyu Chen","submitted_at":"2024-09-26T06:18:06Z","abstract_excerpt":"Software testing is a crucial phase in the software life cycle, helping identify potential risks and reduce maintenance costs. With the advancement of Large Language Models (LLMs), researchers have proposed an increasing number of LLM-based software testing techniques, particularly in the area of test case generation. Despite the growing interest, limited efforts have been made to thoroughly evaluate the actual capabilities of LLMs in this task.\n  In this paper, we introduce TestBench, a benchmark for class-level LLM-based test case generation. We construct a dataset of 108 Java programs from "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.17561","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.17561/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.17561","created_at":"2026-07-05T09:12:11.905247+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.17561v1","created_at":"2026-07-05T09:12:11.905247+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.17561","created_at":"2026-07-05T09:12:11.905247+00:00"},{"alias_kind":"pith_short_12","alias_value":"ED5MISNOIMUP","created_at":"2026-07-05T09:12:11.905247+00:00"},{"alias_kind":"pith_short_16","alias_value":"ED5MISNOIMUPTYO5","created_at":"2026-07-05T09:12:11.905247+00:00"},{"alias_kind":"pith_short_8","alias_value":"ED5MISNO","created_at":"2026-07-05T09:12:11.905247+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2506.02954","citing_title":"Mutation-Guided Unit Test Generation with a Large Language Model","ref_index":74,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01264","citing_title":"FeedbackLLM: Metadata driven Multi-Agentic Language Agnostic Test Case Generator with Evolving prompt and Coverage Feedback","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00942","citing_title":"PPO guided Agentic Pipeline for Adaptive Prompt Selection and Test Case Generation","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22046","citing_title":"Call-Chain-Aware LLM-Based Test Generation for Java Projects","ref_index":44,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ED5MISNOIMUPTYO5K3PQYLJ5OW","json":"https://pith.science/pith/ED5MISNOIMUPTYO5K3PQYLJ5OW.json","graph_json":"https://pith.science/api/pith-number/ED5MISNOIMUPTYO5K3PQYLJ5OW/graph.json","events_json":"https://pith.science/api/pith-number/ED5MISNOIMUPTYO5K3PQYLJ5OW/events.json","paper":"https://pith.science/paper/ED5MISNO"},"agent_actions":{"view_html":"https://pith.science/pith/ED5MISNOIMUPTYO5K3PQYLJ5OW","download_json":"https://pith.science/pith/ED5MISNOIMUPTYO5K3PQYLJ5OW.json","view_paper":"https://pith.science/paper/ED5MISNO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.17561&json=true","fetch_graph":"https://pith.science/api/pith-number/ED5MISNOIMUPTYO5K3PQYLJ5OW/graph.json","fetch_events":"https://pith.science/api/pith-number/ED5MISNOIMUPTYO5K3PQYLJ5OW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ED5MISNOIMUPTYO5K3PQYLJ5OW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ED5MISNOIMUPTYO5K3PQYLJ5OW/action/storage_attestation","attest_author":"https://pith.science/pith/ED5MISNOIMUPTYO5K3PQYLJ5OW/action/author_attestation","sign_citation":"https://pith.science/pith/ED5MISNOIMUPTYO5K3PQYLJ5OW/action/citation_signature","submit_replication":"https://pith.science/pith/ED5MISNOIMUPTYO5K3PQYLJ5OW/action/replication_record"}},"created_at":"2026-07-05T09:12:11.905247+00:00","updated_at":"2026-07-05T09:12:11.905247+00:00"}