{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:PN3ETTHGXVM4RDGTURK5GOGKGC","short_pith_number":"pith:PN3ETTHG","schema_version":"1.0","canonical_sha256":"7b7649cce6bd59c88cd3a455d338ca30bcb66a3a8549fe0596e11943eb496065","source":{"kind":"arxiv","id":"2501.16857","version":1},"attestation_state":"computed","paper":{"title":"Comparing Human and LLM Generated Code: The Jury is Still Out!","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Ansh Bajpai, Chetan Arora, Fanyu Wang, Kla Tantithamthavorn, Sherlock A. Licorish","submitted_at":"2025-01-28T11:11:36Z","abstract_excerpt":"Much is promised in relation to AI-supported software development. However, there has been limited evaluation effort in the research domain aimed at validating the true utility of such techniques, especially when compared to human coding outputs. We bridge this gap, where a benchmark dataset comprising 72 distinct software engineering tasks is used to compare the effectiveness of large language models (LLMs) and human programmers in producing Python software code. GPT-4 is used as a representative LLM, where for the code generated by humans and this LLM, we evaluate code quality and adherence "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.16857","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.SE","submitted_at":"2025-01-28T11:11:36Z","cross_cats_sorted":[],"title_canon_sha256":"6366bd2ec7146b057124b95c6dcf75df114a10824e8106016d7814547e1e10b0","abstract_canon_sha256":"e3987f9d47a004be07205e3e24e6de8bda7ad0ac57af0eb6ee498a0bdef98db1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:06:19.628856Z","signature_b64":"uHWhgP3Tgu0Cj+JoMs2zwT4YjGT8KDuNW4WeZ2f9dJ7m9bE0QLIzWScfvQb5FpGx6fDh2AsU0xBo3jw6KApdAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7b7649cce6bd59c88cd3a455d338ca30bcb66a3a8549fe0596e11943eb496065","last_reissued_at":"2026-07-05T10:06:19.628480Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:06:19.628480Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Comparing Human and LLM Generated Code: The Jury is Still Out!","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Ansh Bajpai, Chetan Arora, Fanyu Wang, Kla Tantithamthavorn, Sherlock A. Licorish","submitted_at":"2025-01-28T11:11:36Z","abstract_excerpt":"Much is promised in relation to AI-supported software development. However, there has been limited evaluation effort in the research domain aimed at validating the true utility of such techniques, especially when compared to human coding outputs. We bridge this gap, where a benchmark dataset comprising 72 distinct software engineering tasks is used to compare the effectiveness of large language models (LLMs) and human programmers in producing Python software code. GPT-4 is used as a representative LLM, where for the code generated by humans and this LLM, we evaluate code quality and adherence "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.16857","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.16857/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.16857","created_at":"2026-07-05T10:06:19.628537+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.16857v1","created_at":"2026-07-05T10:06:19.628537+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.16857","created_at":"2026-07-05T10:06:19.628537+00:00"},{"alias_kind":"pith_short_12","alias_value":"PN3ETTHGXVM4","created_at":"2026-07-05T10:06:19.628537+00:00"},{"alias_kind":"pith_short_16","alias_value":"PN3ETTHGXVM4RDGT","created_at":"2026-07-05T10:06:19.628537+00:00"},{"alias_kind":"pith_short_8","alias_value":"PN3ETTHG","created_at":"2026-07-05T10:06:19.628537+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.04781","citing_title":"AIP: A Graph Representation for Learning and Governing Agent Skills","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00186","citing_title":"How to Compare the Security of Code Written by Humans to LLM-generated Code","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00186","citing_title":"How to Compare the Security of Code Written by Humans to LLM-generated Code","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22349","citing_title":"At What Cost? Software Developers' Well-Being in the Age of GenAI","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19901","citing_title":"Can LLMs Produce Better Object-Oriented Designs than Human-Involved Development?","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13280","citing_title":"Characterizing Readability Issue Patterns and the Role of Prompt Design in LLM-Generated Code","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06464","citing_title":"To What Extent Does Agent-generated Code Require Maintenance? An Empirical Study","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06464","citing_title":"To What Extent Does Agent-generated Code Require Maintenance? An Empirical Study","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09409","citing_title":"Do AI Coding Agents Log Like Humans? An Empirical Study","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PN3ETTHGXVM4RDGTURK5GOGKGC","json":"https://pith.science/pith/PN3ETTHGXVM4RDGTURK5GOGKGC.json","graph_json":"https://pith.science/api/pith-number/PN3ETTHGXVM4RDGTURK5GOGKGC/graph.json","events_json":"https://pith.science/api/pith-number/PN3ETTHGXVM4RDGTURK5GOGKGC/events.json","paper":"https://pith.science/paper/PN3ETTHG"},"agent_actions":{"view_html":"https://pith.science/pith/PN3ETTHGXVM4RDGTURK5GOGKGC","download_json":"https://pith.science/pith/PN3ETTHGXVM4RDGTURK5GOGKGC.json","view_paper":"https://pith.science/paper/PN3ETTHG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.16857&json=true","fetch_graph":"https://pith.science/api/pith-number/PN3ETTHGXVM4RDGTURK5GOGKGC/graph.json","fetch_events":"https://pith.science/api/pith-number/PN3ETTHGXVM4RDGTURK5GOGKGC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PN3ETTHGXVM4RDGTURK5GOGKGC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PN3ETTHGXVM4RDGTURK5GOGKGC/action/storage_attestation","attest_author":"https://pith.science/pith/PN3ETTHGXVM4RDGTURK5GOGKGC/action/author_attestation","sign_citation":"https://pith.science/pith/PN3ETTHGXVM4RDGTURK5GOGKGC/action/citation_signature","submit_replication":"https://pith.science/pith/PN3ETTHGXVM4RDGTURK5GOGKGC/action/replication_record"}},"created_at":"2026-07-05T10:06:19.628537+00:00","updated_at":"2026-07-05T10:06:19.628537+00:00"}