{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:DVDP5LLJZXOGXOXX6VU5F2BDI3","short_pith_number":"pith:DVDP5LLJ","schema_version":"1.0","canonical_sha256":"1d46fead69cddc6bbaf7f569d2e82346e9d1baa3d77168927b5c117c5c43bcf7","source":{"kind":"arxiv","id":"2305.00418","version":4},"attestation_state":"computed","paper":{"title":"Using Large Language Models to Generate JUnit Tests: An Empirical Study","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.SE","authors_text":"Fahmid Al Rifat, Joanna C. S. Santos, Mohammed Latif Siddiq, Noshin Ulfat, Ridwanul Hasan Tanvir, Vinicius Carvalho Lopes","submitted_at":"2023-04-30T07:28:06Z","abstract_excerpt":"A code generation model generates code by taking a prompt from a code comment, existing code, or a combination of both. Although code generation models (e.g., GitHub Copilot) are increasingly being adopted in practice, it is unclear whether they can successfully be used for unit test generation without fine-tuning for a strongly typed language like Java. To fill this gap, we investigated how well three models (Codex, GPT-3.5-Turbo, and StarCoder) can generate unit tests. We used two benchmarks (HumanEval and Evosuite SF110) to investigate the effect of context generation on the unit test gener"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.00418","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SE","submitted_at":"2023-04-30T07:28:06Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"a7832733f7111c6ec3b3144e53f9370d15a4a0e126b6f6b989a01066155936c4","abstract_canon_sha256":"6767a74bd460714cf901b751b8a86a9d2a6441dbc6ef6d9c64fb1ada04b6579c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:59:58.475104Z","signature_b64":"g7W/SFkPXaZCQIcH8zdemuqSwd9VfvlJObOlpghUxCnBj9TWzCls+HPrZF/iaoTITBEVrPSk7UNT9YnTakNbCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1d46fead69cddc6bbaf7f569d2e82346e9d1baa3d77168927b5c117c5c43bcf7","last_reissued_at":"2026-07-05T08:59:58.474675Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:59:58.474675Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Using Large Language Models to Generate JUnit Tests: An Empirical Study","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.SE","authors_text":"Fahmid Al Rifat, Joanna C. S. Santos, Mohammed Latif Siddiq, Noshin Ulfat, Ridwanul Hasan Tanvir, Vinicius Carvalho Lopes","submitted_at":"2023-04-30T07:28:06Z","abstract_excerpt":"A code generation model generates code by taking a prompt from a code comment, existing code, or a combination of both. Although code generation models (e.g., GitHub Copilot) are increasingly being adopted in practice, it is unclear whether they can successfully be used for unit test generation without fine-tuning for a strongly typed language like Java. To fill this gap, we investigated how well three models (Codex, GPT-3.5-Turbo, and StarCoder) can generate unit tests. We used two benchmarks (HumanEval and Evosuite SF110) to investigate the effect of context generation on the unit test gener"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.00418","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.00418/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.00418","created_at":"2026-07-05T08:59:58.474732+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.00418v4","created_at":"2026-07-05T08:59:58.474732+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.00418","created_at":"2026-07-05T08:59:58.474732+00:00"},{"alias_kind":"pith_short_12","alias_value":"DVDP5LLJZXOG","created_at":"2026-07-05T08:59:58.474732+00:00"},{"alias_kind":"pith_short_16","alias_value":"DVDP5LLJZXOGXOXX","created_at":"2026-07-05T08:59:58.474732+00:00"},{"alias_kind":"pith_short_8","alias_value":"DVDP5LLJ","created_at":"2026-07-05T08:59:58.474732+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2505.07700","citing_title":"PatchTrack: A Comprehensive Analysis of ChatGPT's Influence on Pull Request Outcomes","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2403.07974","citing_title":"LiveCodeBench: Holistic and Contamination Free Evaluation of Large Language Models for Code","ref_index":206,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DVDP5LLJZXOGXOXX6VU5F2BDI3","json":"https://pith.science/pith/DVDP5LLJZXOGXOXX6VU5F2BDI3.json","graph_json":"https://pith.science/api/pith-number/DVDP5LLJZXOGXOXX6VU5F2BDI3/graph.json","events_json":"https://pith.science/api/pith-number/DVDP5LLJZXOGXOXX6VU5F2BDI3/events.json","paper":"https://pith.science/paper/DVDP5LLJ"},"agent_actions":{"view_html":"https://pith.science/pith/DVDP5LLJZXOGXOXX6VU5F2BDI3","download_json":"https://pith.science/pith/DVDP5LLJZXOGXOXX6VU5F2BDI3.json","view_paper":"https://pith.science/paper/DVDP5LLJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.00418&json=true","fetch_graph":"https://pith.science/api/pith-number/DVDP5LLJZXOGXOXX6VU5F2BDI3/graph.json","fetch_events":"https://pith.science/api/pith-number/DVDP5LLJZXOGXOXX6VU5F2BDI3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DVDP5LLJZXOGXOXX6VU5F2BDI3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DVDP5LLJZXOGXOXX6VU5F2BDI3/action/storage_attestation","attest_author":"https://pith.science/pith/DVDP5LLJZXOGXOXX6VU5F2BDI3/action/author_attestation","sign_citation":"https://pith.science/pith/DVDP5LLJZXOGXOXX6VU5F2BDI3/action/citation_signature","submit_replication":"https://pith.science/pith/DVDP5LLJZXOGXOXX6VU5F2BDI3/action/replication_record"}},"created_at":"2026-07-05T08:59:58.474732+00:00","updated_at":"2026-07-05T08:59:58.474732+00:00"}