{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:JHFJR7JE2FMYULOAX3QPAGW2M4","short_pith_number":"pith:JHFJR7JE","schema_version":"1.0","canonical_sha256":"49ca98fd24d1598a2dc0bee0f01ada672629d451bc49cf80feac26a3d3c831a0","source":{"kind":"arxiv","id":"2506.07594","version":1},"attestation_state":"computed","paper":{"title":"Evaluating LLMs Effectiveness in Detecting and Correcting Test Smells: An Empirical Study","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Eduardo Santana de Almeida, E. G. Santana Jr, Erlon P. Almeida, Iftekhar Ahmed, Jander Pereira Santos Junior, Paulo Anselmo da Mota Silveira Neto","submitted_at":"2025-06-09T09:46:41Z","abstract_excerpt":"Test smells indicate poor development practices in test code, reducing maintainability and reliability. While developers often struggle to prevent or refactor these issues, existing tools focus primarily on detection rather than automated refactoring. Large Language Models (LLMs) have shown strong potential in code understanding and transformation, but their ability to both identify and refactor test smells remains underexplored. We evaluated GPT-4-Turbo, LLaMA 3 70B, and Gemini-1.5 Pro on Python and Java test suites, using PyNose and TsDetect for initial smell detection, followed by LLM-drive"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.07594","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SE","submitted_at":"2025-06-09T09:46:41Z","cross_cats_sorted":[],"title_canon_sha256":"b63edad99f31b7fd0eb5ce43f72296a6e5df19ccdaecb9c0ef8506ee89ce2252","abstract_canon_sha256":"1a2791360a740836f976ecb8d1049c207f9e80b7bca57a0400f7586821cdef72"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:18:30.052858Z","signature_b64":"L/CmU/ULof0Rz/6FHX5THD1VWVzYuHkQQcD9sGTb7+TH7IaSvyCq5lGF83bT5lDbAgSEZHAoZIxBI3VXRLkZBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"49ca98fd24d1598a2dc0bee0f01ada672629d451bc49cf80feac26a3d3c831a0","last_reissued_at":"2026-07-05T11:18:30.052381Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:18:30.052381Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluating LLMs Effectiveness in Detecting and Correcting Test Smells: An Empirical Study","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Eduardo Santana de Almeida, E. G. Santana Jr, Erlon P. Almeida, Iftekhar Ahmed, Jander Pereira Santos Junior, Paulo Anselmo da Mota Silveira Neto","submitted_at":"2025-06-09T09:46:41Z","abstract_excerpt":"Test smells indicate poor development practices in test code, reducing maintainability and reliability. While developers often struggle to prevent or refactor these issues, existing tools focus primarily on detection rather than automated refactoring. Large Language Models (LLMs) have shown strong potential in code understanding and transformation, but their ability to both identify and refactor test smells remains underexplored. We evaluated GPT-4-Turbo, LLaMA 3 70B, and Gemini-1.5 Pro on Python and Java test suites, using PyNose and TsDetect for initial smell detection, followed by LLM-drive"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.07594","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.07594/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.07594","created_at":"2026-07-05T11:18:30.052438+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.07594v1","created_at":"2026-07-05T11:18:30.052438+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.07594","created_at":"2026-07-05T11:18:30.052438+00:00"},{"alias_kind":"pith_short_12","alias_value":"JHFJR7JE2FMY","created_at":"2026-07-05T11:18:30.052438+00:00"},{"alias_kind":"pith_short_16","alias_value":"JHFJR7JE2FMYULOA","created_at":"2026-07-05T11:18:30.052438+00:00"},{"alias_kind":"pith_short_8","alias_value":"JHFJR7JE","created_at":"2026-07-05T11:18:30.052438+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20173","citing_title":"Qiskit Code Migration with LLMs","ref_index":155,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23361","citing_title":"An Empirical Evaluation of Locally Deployed LLMs for Bug Detection in Python Code","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02091","citing_title":"How Compliant Are GitHub Actions Workflows? A Checklist-Based Study with LLM-Assisted Auditing","ref_index":35,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JHFJR7JE2FMYULOAX3QPAGW2M4","json":"https://pith.science/pith/JHFJR7JE2FMYULOAX3QPAGW2M4.json","graph_json":"https://pith.science/api/pith-number/JHFJR7JE2FMYULOAX3QPAGW2M4/graph.json","events_json":"https://pith.science/api/pith-number/JHFJR7JE2FMYULOAX3QPAGW2M4/events.json","paper":"https://pith.science/paper/JHFJR7JE"},"agent_actions":{"view_html":"https://pith.science/pith/JHFJR7JE2FMYULOAX3QPAGW2M4","download_json":"https://pith.science/pith/JHFJR7JE2FMYULOAX3QPAGW2M4.json","view_paper":"https://pith.science/paper/JHFJR7JE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.07594&json=true","fetch_graph":"https://pith.science/api/pith-number/JHFJR7JE2FMYULOAX3QPAGW2M4/graph.json","fetch_events":"https://pith.science/api/pith-number/JHFJR7JE2FMYULOAX3QPAGW2M4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JHFJR7JE2FMYULOAX3QPAGW2M4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JHFJR7JE2FMYULOAX3QPAGW2M4/action/storage_attestation","attest_author":"https://pith.science/pith/JHFJR7JE2FMYULOAX3QPAGW2M4/action/author_attestation","sign_citation":"https://pith.science/pith/JHFJR7JE2FMYULOAX3QPAGW2M4/action/citation_signature","submit_replication":"https://pith.science/pith/JHFJR7JE2FMYULOAX3QPAGW2M4/action/replication_record"}},"created_at":"2026-07-05T11:18:30.052438+00:00","updated_at":"2026-07-05T11:18:30.052438+00:00"}