{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:NXMSC7CR4YU3XWQ4SNJ55ZW3SO","short_pith_number":"pith:NXMSC7CR","schema_version":"1.0","canonical_sha256":"6dd9217c51e629bbda1c9353dee6db93aafb78c37ce6ec420da6e4eaf3cb216c","source":{"kind":"arxiv","id":"2308.02828","version":2},"attestation_state":"computed","paper":{"title":"An Empirical Study of the Non-determinism of ChatGPT in Code Generation","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Jie M. Zhang, Mark Harman, Meng Wang, Shuyin Ouyang","submitted_at":"2023-08-05T09:30:33Z","abstract_excerpt":"There has been a recent explosion of research on Large Language Models (LLMs) for software engineering tasks, in particular code generation. However, results from LLMs can be highly unstable; nondeterministically returning very different codes for the same prompt. Non-determinism is a potential menace to scientific conclusion validity. When non-determinism is high, scientific conclusions simply cannot be relied upon unless researchers change their behaviour to control for it in their empirical analyses. This paper conducts an empirical study to demonstrate that non-determinism is, indeed, high"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.02828","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"cs.SE","submitted_at":"2023-08-05T09:30:33Z","cross_cats_sorted":[],"title_canon_sha256":"cd4f6d9415731fe4d733b313d8d255ad95618e59779f6a272ad0c9c8421139f0","abstract_canon_sha256":"c892c63bb85bab089fbeb18141ebc7080ea4f9509a608a2d28464561951d45c0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:21:37.897882Z","signature_b64":"gWb9b+f824zIN280748suj+Fi0r7Xg87tGRzBddvgDsT9uHdH47sX5y4EdnbRb4cHklklQ4FLIuzYQrENY+/Cw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6dd9217c51e629bbda1c9353dee6db93aafb78c37ce6ec420da6e4eaf3cb216c","last_reissued_at":"2026-07-05T09:21:37.897421Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:21:37.897421Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"An Empirical Study of the Non-determinism of ChatGPT in Code Generation","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Jie M. Zhang, Mark Harman, Meng Wang, Shuyin Ouyang","submitted_at":"2023-08-05T09:30:33Z","abstract_excerpt":"There has been a recent explosion of research on Large Language Models (LLMs) for software engineering tasks, in particular code generation. However, results from LLMs can be highly unstable; nondeterministically returning very different codes for the same prompt. Non-determinism is a potential menace to scientific conclusion validity. When non-determinism is high, scientific conclusions simply cannot be relied upon unless researchers change their behaviour to control for it in their empirical analyses. This paper conducts an empirical study to demonstrate that non-determinism is, indeed, high"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.02828","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.02828/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.02828","created_at":"2026-07-05T09:21:37.897490+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.02828v2","created_at":"2026-07-05T09:21:37.897490+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.02828","created_at":"2026-07-05T09:21:37.897490+00:00"},{"alias_kind":"pith_short_12","alias_value":"NXMSC7CR4YU3","created_at":"2026-07-05T09:21:37.897490+00:00"},{"alias_kind":"pith_short_16","alias_value":"NXMSC7CR4YU3XWQ4","created_at":"2026-07-05T09:21:37.897490+00:00"},{"alias_kind":"pith_short_8","alias_value":"NXMSC7CR","created_at":"2026-07-05T09:21:37.897490+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2402.09664","citing_title":"CodeMind: Evaluating Large Language Models for Code Reasoning","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22411","citing_title":"Introducing Background Temperature to Characterise Hidden Randomness in Large Language Models","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19826","citing_title":"Co-Located Tests, Better AI Code: How Test Syntax Structure Affects Foundation Model Code Generation","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05150","citing_title":"Compiled AI: Deterministic Code Generation for LLM-Based Workflow Automation","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NXMSC7CR4YU3XWQ4SNJ55ZW3SO","json":"https://pith.science/pith/NXMSC7CR4YU3XWQ4SNJ55ZW3SO.json","graph_json":"https://pith.science/api/pith-number/NXMSC7CR4YU3XWQ4SNJ55ZW3SO/graph.json","events_json":"https://pith.science/api/pith-number/NXMSC7CR4YU3XWQ4SNJ55ZW3SO/events.json","paper":"https://pith.science/paper/NXMSC7CR"},"agent_actions":{"view_html":"https://pith.science/pith/NXMSC7CR4YU3XWQ4SNJ55ZW3SO","download_json":"https://pith.science/pith/NXMSC7CR4YU3XWQ4SNJ55ZW3SO.json","view_paper":"https://pith.science/paper/NXMSC7CR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.02828&json=true","fetch_graph":"https://pith.science/api/pith-number/NXMSC7CR4YU3XWQ4SNJ55ZW3SO/graph.json","fetch_events":"https://pith.science/api/pith-number/NXMSC7CR4YU3XWQ4SNJ55ZW3SO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NXMSC7CR4YU3XWQ4SNJ55ZW3SO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NXMSC7CR4YU3XWQ4SNJ55ZW3SO/action/storage_attestation","attest_author":"https://pith.science/pith/NXMSC7CR4YU3XWQ4SNJ55ZW3SO/action/author_attestation","sign_citation":"https://pith.science/pith/NXMSC7CR4YU3XWQ4SNJ55ZW3SO/action/citation_signature","submit_replication":"https://pith.science/pith/NXMSC7CR4YU3XWQ4SNJ55ZW3SO/action/replication_record"}},"created_at":"2026-07-05T09:21:37.897490+00:00","updated_at":"2026-07-05T09:21:37.897490+00:00"}