{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:VPJTMYJUDAYXPM23XBM5K5WKTF","short_pith_number":"pith:VPJTMYJU","schema_version":"1.0","canonical_sha256":"abd3366134183177b35bb859d576ca9953d01fbd39a8588996d05f07f53cd09c","source":{"kind":"arxiv","id":"2607.22880","version":1},"attestation_state":"computed","paper":{"title":"Do Coverage and Mutation Scores of LLM-Generated Test Suites Correlate with Their Effectiveness? (Replicability Study)","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.SE","authors_text":"Eldan Cohen, Junda Zhao, Shurui Zhou","submitted_at":"2026-07-24T19:46:30Z","abstract_excerpt":"Recent advances in large language models (LLMs) have driven growing interest in using LLMs to automate test generation. Prior work commonly evaluates generated test suites using proxy metrics such as code coverage and mutation score. However, studies by Inozemtseva et al. and Papadakis et al. show that, for human-written tests, correlations among coverage, mutation, and real-bug detection can largely vanish once test suite size is controlled, raising concerns about the validity of evaluations based on proxy metrics. It also remains unclear whether these conclusions carry over to LLM-generated "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.22880","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SE","submitted_at":"2026-07-24T19:46:30Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"2188dd3a5bd083ad67c72fa44d0b1bdb5890ffc2dcb4f0a037e51dda54111752","abstract_canon_sha256":"3f258ce07ac69b8168aa8ddc1ef6f37b8c93dd162f2873104f82ff6649a10155"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-28T00:21:59.927990Z","signature_b64":"j5gCMTFL9aroZbo0JcBpZYzfjLBHdteSh6umlYy7kg4WOGxtkYfGdFp7q21pWNA6tcvwi3lIxPE95dRrK0hACQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"abd3366134183177b35bb859d576ca9953d01fbd39a8588996d05f07f53cd09c","last_reissued_at":"2026-07-28T00:21:59.927096Z","signature_status":"signed_v1","first_computed_at":"2026-07-28T00:21:59.927096Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Do Coverage and Mutation Scores of LLM-Generated Test Suites Correlate with Their Effectiveness? (Replicability Study)","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.SE","authors_text":"Eldan Cohen, Junda Zhao, Shurui Zhou","submitted_at":"2026-07-24T19:46:30Z","abstract_excerpt":"Recent advances in large language models (LLMs) have driven growing interest in using LLMs to automate test generation. Prior work commonly evaluates generated test suites using proxy metrics such as code coverage and mutation score. However, studies by Inozemtseva et al. and Papadakis et al. show that, for human-written tests, correlations among coverage, mutation, and real-bug detection can largely vanish once test suite size is controlled, raising concerns about the validity of evaluations based on proxy metrics. It also remains unclear whether these conclusions carry over to LLM-generated "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.22880","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.22880/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.22880","created_at":"2026-07-28T00:21:59.927573+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.22880v1","created_at":"2026-07-28T00:21:59.927573+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.22880","created_at":"2026-07-28T00:21:59.927573+00:00"},{"alias_kind":"pith_short_12","alias_value":"VPJTMYJUDAYX","created_at":"2026-07-28T00:21:59.927573+00:00"},{"alias_kind":"pith_short_16","alias_value":"VPJTMYJUDAYXPM23","created_at":"2026-07-28T00:21:59.927573+00:00"},{"alias_kind":"pith_short_8","alias_value":"VPJTMYJU","created_at":"2026-07-28T00:21:59.927573+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VPJTMYJUDAYXPM23XBM5K5WKTF","json":"https://pith.science/pith/VPJTMYJUDAYXPM23XBM5K5WKTF.json","graph_json":"https://pith.science/api/pith-number/VPJTMYJUDAYXPM23XBM5K5WKTF/graph.json","events_json":"https://pith.science/api/pith-number/VPJTMYJUDAYXPM23XBM5K5WKTF/events.json","paper":"https://pith.science/paper/VPJTMYJU"},"agent_actions":{"view_html":"https://pith.science/pith/VPJTMYJUDAYXPM23XBM5K5WKTF","download_json":"https://pith.science/pith/VPJTMYJUDAYXPM23XBM5K5WKTF.json","view_paper":"https://pith.science/paper/VPJTMYJU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.22880&json=true","fetch_graph":"https://pith.science/api/pith-number/VPJTMYJUDAYXPM23XBM5K5WKTF/graph.json","fetch_events":"https://pith.science/api/pith-number/VPJTMYJUDAYXPM23XBM5K5WKTF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VPJTMYJUDAYXPM23XBM5K5WKTF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VPJTMYJUDAYXPM23XBM5K5WKTF/action/storage_attestation","attest_author":"https://pith.science/pith/VPJTMYJUDAYXPM23XBM5K5WKTF/action/author_attestation","sign_citation":"https://pith.science/pith/VPJTMYJUDAYXPM23XBM5K5WKTF/action/citation_signature","submit_replication":"https://pith.science/pith/VPJTMYJUDAYXPM23XBM5K5WKTF/action/replication_record"}},"created_at":"2026-07-28T00:21:59.927573+00:00","updated_at":"2026-07-28T00:21:59.927573+00:00"}