{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:DBUAFZWCPEYCBGYP5MNU3VIVWH","short_pith_number":"pith:DBUAFZWC","schema_version":"1.0","canonical_sha256":"186802e6c27930209b0feb1b4dd515b1cd94f8276716b1f32ebd1580881aa1d3","source":{"kind":"arxiv","id":"2406.00215","version":3},"attestation_state":"computed","paper":{"title":"HumanEvalComm: Benchmarking the Communication Competence of Code Generation for LLMs and LLM Agent","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Fatemeh H Fard, Jie JW Wu","submitted_at":"2024-05-31T22:06:18Z","abstract_excerpt":"Large language models (LLMs) have significantly improved their ability to perform tasks in the field of code generation. However, there is still a gap between LLMs being capable coders and being top-tier software engineers. Based on the observation that top-level software engineers often ask clarifying questions to reduce ambiguity in both requirements and coding solutions, we argue that the same should be applied to LLMs for code generation tasks.\n  In this work, we conducted an empirical study on the benchmark and analysis of the communication skills of LLMs for code generation. We define co"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.00215","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SE","submitted_at":"2024-05-31T22:06:18Z","cross_cats_sorted":[],"title_canon_sha256":"30f71e5f45fa85d9eead35db746af48fded5ca5b35c626fc3f4ee3cb38d68dd7","abstract_canon_sha256":"f021719e2d845acbb5b6df9b703298a0d77d045d8976e4a37bfce738cddaea0d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:06:04.195519Z","signature_b64":"c1CI0xkzyg8HIOYirQS0Vv6AQ5CpOBTRnRAdl/zKObSQ6A6cHpNOZW/P+ON6282HeQOc5LRvQObkv5+7KzvYDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"186802e6c27930209b0feb1b4dd515b1cd94f8276716b1f32ebd1580881aa1d3","last_reissued_at":"2026-07-05T10:06:04.195018Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:06:04.195018Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HumanEvalComm: Benchmarking the Communication Competence of Code Generation for LLMs and LLM Agent","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Fatemeh H Fard, Jie JW Wu","submitted_at":"2024-05-31T22:06:18Z","abstract_excerpt":"Large language models (LLMs) have significantly improved their ability to perform tasks in the field of code generation. However, there is still a gap between LLMs being capable coders and being top-tier software engineers. Based on the observation that top-level software engineers often ask clarifying questions to reduce ambiguity in both requirements and coding solutions, we argue that the same should be applied to LLMs for code generation tasks.\n  In this work, we conducted an empirical study on the benchmark and analysis of the communication skills of LLMs for code generation. We define co"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.00215","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.00215/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.00215","created_at":"2026-07-05T10:06:04.195076+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.00215v3","created_at":"2026-07-05T10:06:04.195076+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.00215","created_at":"2026-07-05T10:06:04.195076+00:00"},{"alias_kind":"pith_short_12","alias_value":"DBUAFZWCPEYC","created_at":"2026-07-05T10:06:04.195076+00:00"},{"alias_kind":"pith_short_16","alias_value":"DBUAFZWCPEYCBGYP","created_at":"2026-07-05T10:06:04.195076+00:00"},{"alias_kind":"pith_short_8","alias_value":"DBUAFZWC","created_at":"2026-07-05T10:06:04.195076+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.00107","citing_title":"The Illusion of Safety: Multi-Tier Verification of AI vs. Human C++ Code","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18073","citing_title":"A-ProS: Towards Reliable Autonomous Programming Through Multi-Model Feedback","ref_index":82,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21505","citing_title":"Assessing the Impact of Requirement Ambiguity on LLM-based Function-Level Code Generation","ref_index":40,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DBUAFZWCPEYCBGYP5MNU3VIVWH","json":"https://pith.science/pith/DBUAFZWCPEYCBGYP5MNU3VIVWH.json","graph_json":"https://pith.science/api/pith-number/DBUAFZWCPEYCBGYP5MNU3VIVWH/graph.json","events_json":"https://pith.science/api/pith-number/DBUAFZWCPEYCBGYP5MNU3VIVWH/events.json","paper":"https://pith.science/paper/DBUAFZWC"},"agent_actions":{"view_html":"https://pith.science/pith/DBUAFZWCPEYCBGYP5MNU3VIVWH","download_json":"https://pith.science/pith/DBUAFZWCPEYCBGYP5MNU3VIVWH.json","view_paper":"https://pith.science/paper/DBUAFZWC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.00215&json=true","fetch_graph":"https://pith.science/api/pith-number/DBUAFZWCPEYCBGYP5MNU3VIVWH/graph.json","fetch_events":"https://pith.science/api/pith-number/DBUAFZWCPEYCBGYP5MNU3VIVWH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DBUAFZWCPEYCBGYP5MNU3VIVWH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DBUAFZWCPEYCBGYP5MNU3VIVWH/action/storage_attestation","attest_author":"https://pith.science/pith/DBUAFZWCPEYCBGYP5MNU3VIVWH/action/author_attestation","sign_citation":"https://pith.science/pith/DBUAFZWCPEYCBGYP5MNU3VIVWH/action/citation_signature","submit_replication":"https://pith.science/pith/DBUAFZWCPEYCBGYP5MNU3VIVWH/action/replication_record"}},"created_at":"2026-07-05T10:06:04.195076+00:00","updated_at":"2026-07-05T10:06:04.195076+00:00"}