{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:C4E35LTMQJGSX4HBTZKS73MUJS","short_pith_number":"pith:C4E35LTM","schema_version":"1.0","canonical_sha256":"1709beae6c824d2bf0e19e552fed944c805536675db0713bd565e82f73196909","source":{"kind":"arxiv","id":"2508.00408","version":1},"attestation_state":"computed","paper":{"title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.SE","authors_text":"Dong Huang, Jie M. Zhang, Mark Harman, Mingzhe Du, Qianru Zhang, See-kiong Ng","submitted_at":"2025-08-01T08:08:26Z","abstract_excerpt":"Recently, large language models (LLMs) have shown great promise in automating unit test generation, significantly reducing the manual effort required by developers. To effectively evaluate the capabilities of LLMs in this domain, it is crucial to have a well-designed benchmark that accurately reflects real-world scenarios and mitigates common pitfalls. Existing LLM test generation benchmarks are limited by two critical drawbacks: data contamination and structurally simple function code. As a result, we often cannot rely on the validity of scientific conclusions drawn from empirical studies usi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.00408","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SE","submitted_at":"2025-08-01T08:08:26Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"5297da1be6495b87cd339480c4c01401218d5522c7cb43764ab7b674f37c08e1","abstract_canon_sha256":"e18b662deca4048e460893b2c92adcea214616940dd7755a0aac9b97418efd0f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:46:55.488078Z","signature_b64":"ai3ZwR5oB97jRYeD18ki4US1sNmVMEui4Q7A5FsS1m1PhMS87pTe62l7LOCjAhjmuULyv6+rPfRQ0gW03/ZzBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1709beae6c824d2bf0e19e552fed944c805536675db0713bd565e82f73196909","last_reissued_at":"2026-07-05T11:46:55.487606Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:46:55.487606Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Benchmarking LLMs for Unit Test Generation from Real-World Functions","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.SE","authors_text":"Dong Huang, Jie M. Zhang, Mark Harman, Mingzhe Du, Qianru Zhang, See-kiong Ng","submitted_at":"2025-08-01T08:08:26Z","abstract_excerpt":"Recently, large language models (LLMs) have shown great promise in automating unit test generation, significantly reducing the manual effort required by developers. To effectively evaluate the capabilities of LLMs in this domain, it is crucial to have a well-designed benchmark that accurately reflects real-world scenarios and mitigates common pitfalls. Existing LLM test generation benchmarks are limited by two critical drawbacks: data contamination and structurally simple function code. As a result, we often cannot rely on the validity of scientific conclusions drawn from empirical studies usi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.00408","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.00408/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.00408","created_at":"2026-07-05T11:46:55.487664+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.00408v1","created_at":"2026-07-05T11:46:55.487664+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.00408","created_at":"2026-07-05T11:46:55.487664+00:00"},{"alias_kind":"pith_short_12","alias_value":"C4E35LTMQJGS","created_at":"2026-07-05T11:46:55.487664+00:00"},{"alias_kind":"pith_short_16","alias_value":"C4E35LTMQJGSX4HB","created_at":"2026-07-05T11:46:55.487664+00:00"},{"alias_kind":"pith_short_8","alias_value":"C4E35LTM","created_at":"2026-07-05T11:46:55.487664+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.01799","citing_title":"TestDecision: Sequential Test Suite Generation via Greedy Optimization and Reinforcement Learning","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/C4E35LTMQJGSX4HBTZKS73MUJS","json":"https://pith.science/pith/C4E35LTMQJGSX4HBTZKS73MUJS.json","graph_json":"https://pith.science/api/pith-number/C4E35LTMQJGSX4HBTZKS73MUJS/graph.json","events_json":"https://pith.science/api/pith-number/C4E35LTMQJGSX4HBTZKS73MUJS/events.json","paper":"https://pith.science/paper/C4E35LTM"},"agent_actions":{"view_html":"https://pith.science/pith/C4E35LTMQJGSX4HBTZKS73MUJS","download_json":"https://pith.science/pith/C4E35LTMQJGSX4HBTZKS73MUJS.json","view_paper":"https://pith.science/paper/C4E35LTM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.00408&json=true","fetch_graph":"https://pith.science/api/pith-number/C4E35LTMQJGSX4HBTZKS73MUJS/graph.json","fetch_events":"https://pith.science/api/pith-number/C4E35LTMQJGSX4HBTZKS73MUJS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/C4E35LTMQJGSX4HBTZKS73MUJS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/C4E35LTMQJGSX4HBTZKS73MUJS/action/storage_attestation","attest_author":"https://pith.science/pith/C4E35LTMQJGSX4HBTZKS73MUJS/action/author_attestation","sign_citation":"https://pith.science/pith/C4E35LTMQJGSX4HBTZKS73MUJS/action/citation_signature","submit_replication":"https://pith.science/pith/C4E35LTMQJGSX4HBTZKS73MUJS/action/replication_record"}},"created_at":"2026-07-05T11:46:55.487664+00:00","updated_at":"2026-07-05T11:46:55.487664+00:00"}