{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:BVG66P45M2NBRVDQNVCAHJPSRB","short_pith_number":"pith:BVG66P45","schema_version":"1.0","canonical_sha256":"0d4def3f9d669a18d4706d4403a5f28862b23976b1844fe2eccef494949c8785","source":{"kind":"arxiv","id":"2409.09464","version":3},"attestation_state":"computed","paper":{"title":"Measuring the Influence of Incorrect Code on Test Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.SE","authors_text":"Dong Huang, Heming Cui, Jie M. Zhang, Mark Harman, Mingzhe Du","submitted_at":"2024-09-14T15:17:34Z","abstract_excerpt":"It is natural to suppose that a Large Language Model is more likely to generate correct test cases when prompted with correct code under test, compared to incorrect code under test. However, the size of this effect has never been previously measured, despite its obvious importance for both practicing software engineers and researchers. To answer the question, we conducted a comprehensive empirical study on 5 open source and 6 closed source language models, with 3 widely-used benchmark data sets together with 41 repo-level real-world examples from two different real-world data sets. Our results"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.09464","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SE","submitted_at":"2024-09-14T15:17:34Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"bd9cb061e35fe6fab3d1d4c0ab0bd3b19c98af352013336dd9b50c07c8e0a26c","abstract_canon_sha256":"fe1b3f9d3774338b075cd6102b6407af46af161e8cbfa4f515ae35ef1064f500"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:40:35.629741Z","signature_b64":"qdp6EcdDJAE6uZboveaEzYyd1wGeH34JeBzeNQvN4nE4ZJ9de+TIabvKElA6INYbF2PXVvzg6cRGUfFlpWPpAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0d4def3f9d669a18d4706d4403a5f28862b23976b1844fe2eccef494949c8785","last_reissued_at":"2026-07-05T10:40:35.629349Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:40:35.629349Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Measuring the Influence of Incorrect Code on Test Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.SE","authors_text":"Dong Huang, Heming Cui, Jie M. Zhang, Mark Harman, Mingzhe Du","submitted_at":"2024-09-14T15:17:34Z","abstract_excerpt":"It is natural to suppose that a Large Language Model is more likely to generate correct test cases when prompted with correct code under test, compared to incorrect code under test. However, the size of this effect has never been previously measured, despite its obvious importance for both practicing software engineers and researchers. To answer the question, we conducted a comprehensive empirical study on 5 open source and 6 closed source language models, with 3 widely-used benchmark data sets together with 41 repo-level real-world examples from two different real-world data sets. Our results"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.09464","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.09464/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.09464","created_at":"2026-07-05T10:40:35.629406+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.09464v3","created_at":"2026-07-05T10:40:35.629406+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.09464","created_at":"2026-07-05T10:40:35.629406+00:00"},{"alias_kind":"pith_short_12","alias_value":"BVG66P45M2NB","created_at":"2026-07-05T10:40:35.629406+00:00"},{"alias_kind":"pith_short_16","alias_value":"BVG66P45M2NBRVDQ","created_at":"2026-07-05T10:40:35.629406+00:00"},{"alias_kind":"pith_short_8","alias_value":"BVG66P45","created_at":"2026-07-05T10:40:35.629406+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.29822","citing_title":"Inferring Code Correctness from Specification","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BVG66P45M2NBRVDQNVCAHJPSRB","json":"https://pith.science/pith/BVG66P45M2NBRVDQNVCAHJPSRB.json","graph_json":"https://pith.science/api/pith-number/BVG66P45M2NBRVDQNVCAHJPSRB/graph.json","events_json":"https://pith.science/api/pith-number/BVG66P45M2NBRVDQNVCAHJPSRB/events.json","paper":"https://pith.science/paper/BVG66P45"},"agent_actions":{"view_html":"https://pith.science/pith/BVG66P45M2NBRVDQNVCAHJPSRB","download_json":"https://pith.science/pith/BVG66P45M2NBRVDQNVCAHJPSRB.json","view_paper":"https://pith.science/paper/BVG66P45","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.09464&json=true","fetch_graph":"https://pith.science/api/pith-number/BVG66P45M2NBRVDQNVCAHJPSRB/graph.json","fetch_events":"https://pith.science/api/pith-number/BVG66P45M2NBRVDQNVCAHJPSRB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BVG66P45M2NBRVDQNVCAHJPSRB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BVG66P45M2NBRVDQNVCAHJPSRB/action/storage_attestation","attest_author":"https://pith.science/pith/BVG66P45M2NBRVDQNVCAHJPSRB/action/author_attestation","sign_citation":"https://pith.science/pith/BVG66P45M2NBRVDQNVCAHJPSRB/action/citation_signature","submit_replication":"https://pith.science/pith/BVG66P45M2NBRVDQNVCAHJPSRB/action/replication_record"}},"created_at":"2026-07-05T10:40:35.629406+00:00","updated_at":"2026-07-05T10:40:35.629406+00:00"}