{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:7L4ACJZA4D7V7VYOLSLS7HISE6","short_pith_number":"pith:7L4ACJZA","schema_version":"1.0","canonical_sha256":"faf8012720e0ff5fd70e5c972f9d1227a76bf3cde2bcdc9d7c789a6ed85d8fcd","source":{"kind":"arxiv","id":"2506.14297","version":1},"attestation_state":"computed","paper":{"title":"Quality Assessment of Python Tests Generated by Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Carla Bezerra, Ivan Machado, Larissa Rocha, Publio Silva, T\\'assio Virg\\'inio, Victor Alves","submitted_at":"2025-06-17T08:16:15Z","abstract_excerpt":"The manual generation of test scripts is a time-intensive, costly, and error-prone process, indicating the value of automated solutions. Large Language Models (LLMs) have shown great promise in this domain, leveraging their extensive knowledge to produce test code more efficiently. This study investigates the quality of Python test code generated by three LLMs: GPT-4o, Amazon Q, and LLama 3.3. We evaluate the structural reliability of test suites generated under two distinct prompt contexts: Text2Code (T2C) and Code2Code (C2C). Our analysis includes the identification of errors and test smells"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.14297","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SE","submitted_at":"2025-06-17T08:16:15Z","cross_cats_sorted":[],"title_canon_sha256":"1a4a197db56aa733f6a5e483061a5000660db6b8202637073170e8e60c9b6c80","abstract_canon_sha256":"6dfbee977c07acc58cbed56582cd9793012491b0a18dbf8c51702ba56d3a4949"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:22:52.858911Z","signature_b64":"8VgGGjrdHrrqhh4cxKv9O5+8CxDIPNhBgj9nxaQiXr1H2Zfw/4U4TgfusujvNFyvIDY2doiWdHpraxcOPe2eDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"faf8012720e0ff5fd70e5c972f9d1227a76bf3cde2bcdc9d7c789a6ed85d8fcd","last_reissued_at":"2026-07-05T11:22:52.858437Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:22:52.858437Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Quality Assessment of Python Tests Generated by Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Carla Bezerra, Ivan Machado, Larissa Rocha, Publio Silva, T\\'assio Virg\\'inio, Victor Alves","submitted_at":"2025-06-17T08:16:15Z","abstract_excerpt":"The manual generation of test scripts is a time-intensive, costly, and error-prone process, indicating the value of automated solutions. Large Language Models (LLMs) have shown great promise in this domain, leveraging their extensive knowledge to produce test code more efficiently. This study investigates the quality of Python test code generated by three LLMs: GPT-4o, Amazon Q, and LLama 3.3. We evaluate the structural reliability of test suites generated under two distinct prompt contexts: Text2Code (T2C) and Code2Code (C2C). Our analysis includes the identification of errors and test smells"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.14297","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.14297/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.14297","created_at":"2026-07-05T11:22:52.858495+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.14297v1","created_at":"2026-07-05T11:22:52.858495+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.14297","created_at":"2026-07-05T11:22:52.858495+00:00"},{"alias_kind":"pith_short_12","alias_value":"7L4ACJZA4D7V","created_at":"2026-07-05T11:22:52.858495+00:00"},{"alias_kind":"pith_short_16","alias_value":"7L4ACJZA4D7V7VYO","created_at":"2026-07-05T11:22:52.858495+00:00"},{"alias_kind":"pith_short_8","alias_value":"7L4ACJZA","created_at":"2026-07-05T11:22:52.858495+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.23340","citing_title":"Can LLMs be Effective Code Contributors? A Study on Open-source Projects","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05267","citing_title":"Bridging Generation and Training: A Systematic Review of Quality Issues in LLMs for Code","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7L4ACJZA4D7V7VYOLSLS7HISE6","json":"https://pith.science/pith/7L4ACJZA4D7V7VYOLSLS7HISE6.json","graph_json":"https://pith.science/api/pith-number/7L4ACJZA4D7V7VYOLSLS7HISE6/graph.json","events_json":"https://pith.science/api/pith-number/7L4ACJZA4D7V7VYOLSLS7HISE6/events.json","paper":"https://pith.science/paper/7L4ACJZA"},"agent_actions":{"view_html":"https://pith.science/pith/7L4ACJZA4D7V7VYOLSLS7HISE6","download_json":"https://pith.science/pith/7L4ACJZA4D7V7VYOLSLS7HISE6.json","view_paper":"https://pith.science/paper/7L4ACJZA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.14297&json=true","fetch_graph":"https://pith.science/api/pith-number/7L4ACJZA4D7V7VYOLSLS7HISE6/graph.json","fetch_events":"https://pith.science/api/pith-number/7L4ACJZA4D7V7VYOLSLS7HISE6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7L4ACJZA4D7V7VYOLSLS7HISE6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7L4ACJZA4D7V7VYOLSLS7HISE6/action/storage_attestation","attest_author":"https://pith.science/pith/7L4ACJZA4D7V7VYOLSLS7HISE6/action/author_attestation","sign_citation":"https://pith.science/pith/7L4ACJZA4D7V7VYOLSLS7HISE6/action/citation_signature","submit_replication":"https://pith.science/pith/7L4ACJZA4D7V7VYOLSLS7HISE6/action/replication_record"}},"created_at":"2026-07-05T11:22:52.858495+00:00","updated_at":"2026-07-05T11:22:52.858495+00:00"}