{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:DW64SC3ZRX2VRMDINFN73KJEX7","short_pith_number":"pith:DW64SC3Z","schema_version":"1.0","canonical_sha256":"1dbdc90b798df558b068695bfda924bfd4747fd9c332bee93717ed8f571a834c","source":{"kind":"arxiv","id":"2311.07599","version":1},"attestation_state":"computed","paper":{"title":"Testing LLMs on Code Generation with Varying Levels of Prompt Specificity","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.SE","authors_text":"David Gao, Lincoln Murr, Morgan Grainger","submitted_at":"2023-11-10T23:41:41Z","abstract_excerpt":"Large language models (LLMs) have demonstrated unparalleled prowess in mimicking human-like text generation and processing. Among the myriad of applications that benefit from LLMs, automated code generation is increasingly promising. The potential to transform natural language prompts into executable code promises a major shift in software development practices and paves the way for significant reductions in manual coding efforts and the likelihood of human-induced errors. This paper reports the results of a study that evaluates the performance of various LLMs, such as Bard, ChatGPT-3.5, ChatG"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.07599","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SE","submitted_at":"2023-11-10T23:41:41Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"0ee3d5b141d8672102e7ce56d4c811ee42ad6f4d4d61acbb7ab803cfd701b762","abstract_canon_sha256":"f3f38199b22a26cd637bcc16f36c257d7f8bc3104764f74c6631c78b942c7159"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:12:20.972572Z","signature_b64":"TBEHHU7KVWjmsPOxkYyLYqT3FPnjuvSksfi2M7AStAEicOzVqx9vsG47fiTVwxJ7F+cJKyjfuokMYfFdSzm/DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1dbdc90b798df558b068695bfda924bfd4747fd9c332bee93717ed8f571a834c","last_reissued_at":"2026-07-05T07:12:20.972195Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:12:20.972195Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Testing LLMs on Code Generation with Varying Levels of Prompt Specificity","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.SE","authors_text":"David Gao, Lincoln Murr, Morgan Grainger","submitted_at":"2023-11-10T23:41:41Z","abstract_excerpt":"Large language models (LLMs) have demonstrated unparalleled prowess in mimicking human-like text generation and processing. Among the myriad of applications that benefit from LLMs, automated code generation is increasingly promising. The potential to transform natural language prompts into executable code promises a major shift in software development practices and paves the way for significant reductions in manual coding efforts and the likelihood of human-induced errors. This paper reports the results of a study that evaluates the performance of various LLMs, such as Bard, ChatGPT-3.5, ChatG"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.07599","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.07599/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.07599","created_at":"2026-07-05T07:12:20.972250+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.07599v1","created_at":"2026-07-05T07:12:20.972250+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.07599","created_at":"2026-07-05T07:12:20.972250+00:00"},{"alias_kind":"pith_short_12","alias_value":"DW64SC3ZRX2V","created_at":"2026-07-05T07:12:20.972250+00:00"},{"alias_kind":"pith_short_16","alias_value":"DW64SC3ZRX2VRMDI","created_at":"2026-07-05T07:12:20.972250+00:00"},{"alias_kind":"pith_short_8","alias_value":"DW64SC3Z","created_at":"2026-07-05T07:12:20.972250+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.14816","citing_title":"Large Language Models for Code Generation from Multilingual Prompts: A Curated Benchmark and a Study on Code Quality","ref_index":65,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DW64SC3ZRX2VRMDINFN73KJEX7","json":"https://pith.science/pith/DW64SC3ZRX2VRMDINFN73KJEX7.json","graph_json":"https://pith.science/api/pith-number/DW64SC3ZRX2VRMDINFN73KJEX7/graph.json","events_json":"https://pith.science/api/pith-number/DW64SC3ZRX2VRMDINFN73KJEX7/events.json","paper":"https://pith.science/paper/DW64SC3Z"},"agent_actions":{"view_html":"https://pith.science/pith/DW64SC3ZRX2VRMDINFN73KJEX7","download_json":"https://pith.science/pith/DW64SC3ZRX2VRMDINFN73KJEX7.json","view_paper":"https://pith.science/paper/DW64SC3Z","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.07599&json=true","fetch_graph":"https://pith.science/api/pith-number/DW64SC3ZRX2VRMDINFN73KJEX7/graph.json","fetch_events":"https://pith.science/api/pith-number/DW64SC3ZRX2VRMDINFN73KJEX7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DW64SC3ZRX2VRMDINFN73KJEX7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DW64SC3ZRX2VRMDINFN73KJEX7/action/storage_attestation","attest_author":"https://pith.science/pith/DW64SC3ZRX2VRMDINFN73KJEX7/action/author_attestation","sign_citation":"https://pith.science/pith/DW64SC3ZRX2VRMDINFN73KJEX7/action/citation_signature","submit_replication":"https://pith.science/pith/DW64SC3ZRX2VRMDINFN73KJEX7/action/replication_record"}},"created_at":"2026-07-05T07:12:20.972250+00:00","updated_at":"2026-07-05T07:12:20.972250+00:00"}