{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:PC3WKVVZN23B5MUVCW3MFYF4NY","short_pith_number":"pith:PC3WKVVZ","schema_version":"1.0","canonical_sha256":"78b76556b96eb61eb29515b6c2e0bc6e1590ca235f5e0e863b7ba16e1fc19a10","source":{"kind":"arxiv","id":"2304.11686","version":6},"attestation_state":"computed","paper":{"title":"Nuances are the Key: Unlocking ChatGPT to Find Failure-Inducing Tests with Differential Prompting","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Haoye Tian, Jeff Kramer, Shing-Chi Cheung, Tsz-On Li, Wenxi Zong, Yibo Wang, Ying Wang","submitted_at":"2023-04-23T15:35:39Z","abstract_excerpt":"Automatically detecting software failures is an important task and a longstanding challenge. It requires finding failure-inducing test cases whose test input can trigger the software's fault, and constructing an automated oracle to detect the software's incorrect behaviors. Recent advancement of large language models (LLMs) motivates us to study how far this challenge can be addressed by ChatGPT, a state-of-the-art LLM. Unfortunately, our study shows that ChatGPT has a low probability (28.8%) of finding correct failure-inducing test cases for buggy programs. A possible reason is that finding f"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2304.11686","kind":"arxiv","version":6},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SE","submitted_at":"2023-04-23T15:35:39Z","cross_cats_sorted":[],"title_canon_sha256":"3b2e97af8912ef290a0af532a2fa04cfab83dadb8f594f7c9d6c62b0b75672cc","abstract_canon_sha256":"e5a0bb1d47a56baa30ac55666826dc7a7fa7267a7a2c90c99d4b28d9b839d8bc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:49:02.272199Z","signature_b64":"xFpVNmjyP4eKVjVJ2+gMFCqFEVxp01RdyxZ3sFjt7PLELM3XdiIFld5/f8pD0evSqIQP53QVz1HKarebJ1MJBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"78b76556b96eb61eb29515b6c2e0bc6e1590ca235f5e0e863b7ba16e1fc19a10","last_reissued_at":"2026-07-05T06:49:02.271703Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:49:02.271703Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Nuances are the Key: Unlocking ChatGPT to Find Failure-Inducing Tests with Differential Prompting","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Haoye Tian, Jeff Kramer, Shing-Chi Cheung, Tsz-On Li, Wenxi Zong, Yibo Wang, Ying Wang","submitted_at":"2023-04-23T15:35:39Z","abstract_excerpt":"Automatically detecting software failures is an important task and a longstanding challenge. It requires finding failure-inducing test cases whose test input can trigger the software's fault, and constructing an automated oracle to detect the software's incorrect behaviors. Recent advancement of large language models (LLMs) motivates us to study how far this challenge can be addressed by ChatGPT, a state-of-the-art LLM. Unfortunately, our study shows that ChatGPT has a low probability (28.8%) of finding correct failure-inducing test cases for buggy programs. A possible reason is that finding f"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2304.11686","kind":"arxiv","version":6},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2304.11686/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2304.11686","created_at":"2026-07-05T06:49:02.271761+00:00"},{"alias_kind":"arxiv_version","alias_value":"2304.11686v6","created_at":"2026-07-05T06:49:02.271761+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2304.11686","created_at":"2026-07-05T06:49:02.271761+00:00"},{"alias_kind":"pith_short_12","alias_value":"PC3WKVVZN23B","created_at":"2026-07-05T06:49:02.271761+00:00"},{"alias_kind":"pith_short_16","alias_value":"PC3WKVVZN23B5MUV","created_at":"2026-07-05T06:49:02.271761+00:00"},{"alias_kind":"pith_short_8","alias_value":"PC3WKVVZ","created_at":"2026-07-05T06:49:02.271761+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.13629","citing_title":"Large Language Models in Cybersecurity: Applications, Vulnerabilities, and Defense Techniques","ref_index":82,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PC3WKVVZN23B5MUVCW3MFYF4NY","json":"https://pith.science/pith/PC3WKVVZN23B5MUVCW3MFYF4NY.json","graph_json":"https://pith.science/api/pith-number/PC3WKVVZN23B5MUVCW3MFYF4NY/graph.json","events_json":"https://pith.science/api/pith-number/PC3WKVVZN23B5MUVCW3MFYF4NY/events.json","paper":"https://pith.science/paper/PC3WKVVZ"},"agent_actions":{"view_html":"https://pith.science/pith/PC3WKVVZN23B5MUVCW3MFYF4NY","download_json":"https://pith.science/pith/PC3WKVVZN23B5MUVCW3MFYF4NY.json","view_paper":"https://pith.science/paper/PC3WKVVZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2304.11686&json=true","fetch_graph":"https://pith.science/api/pith-number/PC3WKVVZN23B5MUVCW3MFYF4NY/graph.json","fetch_events":"https://pith.science/api/pith-number/PC3WKVVZN23B5MUVCW3MFYF4NY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PC3WKVVZN23B5MUVCW3MFYF4NY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PC3WKVVZN23B5MUVCW3MFYF4NY/action/storage_attestation","attest_author":"https://pith.science/pith/PC3WKVVZN23B5MUVCW3MFYF4NY/action/author_attestation","sign_citation":"https://pith.science/pith/PC3WKVVZN23B5MUVCW3MFYF4NY/action/citation_signature","submit_replication":"https://pith.science/pith/PC3WKVVZN23B5MUVCW3MFYF4NY/action/replication_record"}},"created_at":"2026-07-05T06:49:02.271761+00:00","updated_at":"2026-07-05T06:49:02.271761+00:00"}