{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:MPTWST6JU5CATJXFEB6UATCMSE","short_pith_number":"pith:MPTWST6J","schema_version":"1.0","canonical_sha256":"63e7694fc9a74409a6e5207d404c4c913725258760b919c0f762ff81c6a7a739","source":{"kind":"arxiv","id":"2507.20439","version":1},"attestation_state":"computed","paper":{"title":"When Prompts Go Wrong: Evaluating Code Model Robustness to Ambiguous, Contradictory, and Incomplete Task Descriptions","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.SE","authors_text":"Amal Akli, Federica Sarro, Maxime Cordy, Maya Larbi, Mike Papadakis, Rihab Bouyousfi, Yves Le Traon","submitted_at":"2025-07-27T23:16:14Z","abstract_excerpt":"Large Language Models (LLMs) have demonstrated impressive performance in code generation tasks under idealized conditions, where task descriptions are clear and precise. However, in practice, task descriptions frequently exhibit ambiguity, incompleteness, or internal contradictions. In this paper, we present the first empirical study examining the robustness of state-of-the-art code generation models when faced with such unclear task descriptions. We extend the HumanEval and MBPP benchmarks by systematically introducing realistic task descriptions flaws through guided mutation strategies, prod"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.20439","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SE","submitted_at":"2025-07-27T23:16:14Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"edb4d861bef72c0c1b700f497db0494225001ca66cd647acfa400669c3ed99d6","abstract_canon_sha256":"6bc403b950d107a56afa6d5fc586a7f857ab066bcef1316ae55f33e95c29e5a2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:44:19.421891Z","signature_b64":"FbbIwOfxt4wYxuYrTaKm5Sfv0Zo0RE0EuIrEyIxWqP+nf3sUX61PNxA0ND50Adt0q7TF53hOrmpZevQcTJ7wDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"63e7694fc9a74409a6e5207d404c4c913725258760b919c0f762ff81c6a7a739","last_reissued_at":"2026-07-05T11:44:19.421441Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:44:19.421441Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"When Prompts Go Wrong: Evaluating Code Model Robustness to Ambiguous, Contradictory, and Incomplete Task Descriptions","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.SE","authors_text":"Amal Akli, Federica Sarro, Maxime Cordy, Maya Larbi, Mike Papadakis, Rihab Bouyousfi, Yves Le Traon","submitted_at":"2025-07-27T23:16:14Z","abstract_excerpt":"Large Language Models (LLMs) have demonstrated impressive performance in code generation tasks under idealized conditions, where task descriptions are clear and precise. However, in practice, task descriptions frequently exhibit ambiguity, incompleteness, or internal contradictions. In this paper, we present the first empirical study examining the robustness of state-of-the-art code generation models when faced with such unclear task descriptions. We extend the HumanEval and MBPP benchmarks by systematically introducing realistic task descriptions flaws through guided mutation strategies, prod"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.20439","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.20439/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.20439","created_at":"2026-07-05T11:44:19.421499+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.20439v1","created_at":"2026-07-05T11:44:19.421499+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.20439","created_at":"2026-07-05T11:44:19.421499+00:00"},{"alias_kind":"pith_short_12","alias_value":"MPTWST6JU5CA","created_at":"2026-07-05T11:44:19.421499+00:00"},{"alias_kind":"pith_short_16","alias_value":"MPTWST6JU5CATJXF","created_at":"2026-07-05T11:44:19.421499+00:00"},{"alias_kind":"pith_short_8","alias_value":"MPTWST6J","created_at":"2026-07-05T11:44:19.421499+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01953","citing_title":"Underspecification does not imply Incoherence: The Risks of Semantic Collapse in Coding Models","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02333","citing_title":"Guiding Human Validation of LLM-Generated Code via Verifiable Literate Programming","ref_index":79,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00711","citing_title":"ClarifyCodeBench: Evaluating LLMs on Clarifying Ambiguous Requirements for Code Generation","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24712","citing_title":"When Prompt Under-Specification Improves Code Correctness: An Exploratory Study of Prompt Wording and Structure Effects on LLM-Based Code Generation","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24703","citing_title":"Defective Task Descriptions in LLM-Based Code Generation: Detection and Analysis","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MPTWST6JU5CATJXFEB6UATCMSE","json":"https://pith.science/pith/MPTWST6JU5CATJXFEB6UATCMSE.json","graph_json":"https://pith.science/api/pith-number/MPTWST6JU5CATJXFEB6UATCMSE/graph.json","events_json":"https://pith.science/api/pith-number/MPTWST6JU5CATJXFEB6UATCMSE/events.json","paper":"https://pith.science/paper/MPTWST6J"},"agent_actions":{"view_html":"https://pith.science/pith/MPTWST6JU5CATJXFEB6UATCMSE","download_json":"https://pith.science/pith/MPTWST6JU5CATJXFEB6UATCMSE.json","view_paper":"https://pith.science/paper/MPTWST6J","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.20439&json=true","fetch_graph":"https://pith.science/api/pith-number/MPTWST6JU5CATJXFEB6UATCMSE/graph.json","fetch_events":"https://pith.science/api/pith-number/MPTWST6JU5CATJXFEB6UATCMSE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MPTWST6JU5CATJXFEB6UATCMSE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MPTWST6JU5CATJXFEB6UATCMSE/action/storage_attestation","attest_author":"https://pith.science/pith/MPTWST6JU5CATJXFEB6UATCMSE/action/author_attestation","sign_citation":"https://pith.science/pith/MPTWST6JU5CATJXFEB6UATCMSE/action/citation_signature","submit_replication":"https://pith.science/pith/MPTWST6JU5CATJXFEB6UATCMSE/action/replication_record"}},"created_at":"2026-07-05T11:44:19.421499+00:00","updated_at":"2026-07-05T11:44:19.421499+00:00"}