{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:MSXPHTO7UQPTJ6DLYANH445JKS","short_pith_number":"pith:MSXPHTO7","schema_version":"1.0","canonical_sha256":"64aef3cddfa41f34f86bc01a7e73a9549103f3b2826fde2bf26c6a02fb4d67f4","source":{"kind":"arxiv","id":"2304.14334","version":1},"attestation_state":"computed","paper":{"title":"ZeroShotDataAug: Generating and Augmenting Training Data with ChatGPT","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Rodney Nielsen, Solomon Ubani, Suleyman Olcay Polat","submitted_at":"2023-04-27T17:07:29Z","abstract_excerpt":"In this paper, we investigate the use of data obtained from prompting a large generative language model, ChatGPT, to generate synthetic training data with the aim of augmenting data in low resource scenarios. We show that with appropriate task-specific ChatGPT prompts, we outperform the most popular existing approaches for such data augmentation. Furthermore, we investigate methodologies for evaluating the similarity of the augmented data generated from ChatGPT with the aim of validating and assessing the quality of the data generated."},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2304.14334","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2023-04-27T17:07:29Z","cross_cats_sorted":[],"title_canon_sha256":"fecc29fb3ccfec53890d1644d981b86f14ebffb90fd81f596332095d56afd89f","abstract_canon_sha256":"0d33cdad15a5fec17d9ed2bc0b2d39b109391c040816815e37bb123b471f5c3e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:05:03.207845Z","signature_b64":"R61knEJjJJF+qHK7XW2HHaZWZ//hzb2siQeFTEz7ok+hPGMbQ8Tl8R7XeNsl9/qEkqNFre/z9GFuBIjJbgXMAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"64aef3cddfa41f34f86bc01a7e73a9549103f3b2826fde2bf26c6a02fb4d67f4","last_reissued_at":"2026-07-05T06:05:03.207376Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:05:03.207376Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ZeroShotDataAug: Generating and Augmenting Training Data with ChatGPT","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Rodney Nielsen, Solomon Ubani, Suleyman Olcay Polat","submitted_at":"2023-04-27T17:07:29Z","abstract_excerpt":"In this paper, we investigate the use of data obtained from prompting a large generative language model, ChatGPT, to generate synthetic training data with the aim of augmenting data in low resource scenarios. We show that with appropriate task-specific ChatGPT prompts, we outperform the most popular existing approaches for such data augmentation. Furthermore, we investigate methodologies for evaluating the similarity of the augmented data generated from ChatGPT with the aim of validating and assessing the quality of the data generated."},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2304.14334","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2304.14334/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2304.14334","created_at":"2026-07-05T06:05:03.207434+00:00"},{"alias_kind":"arxiv_version","alias_value":"2304.14334v1","created_at":"2026-07-05T06:05:03.207434+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2304.14334","created_at":"2026-07-05T06:05:03.207434+00:00"},{"alias_kind":"pith_short_12","alias_value":"MSXPHTO7UQPT","created_at":"2026-07-05T06:05:03.207434+00:00"},{"alias_kind":"pith_short_16","alias_value":"MSXPHTO7UQPTJ6DL","created_at":"2026-07-05T06:05:03.207434+00:00"},{"alias_kind":"pith_short_8","alias_value":"MSXPHTO7","created_at":"2026-07-05T06:05:03.207434+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08437","citing_title":"Does online sustainability communication shape public discourse? Insights from six years of tenant-housing provider interactions","ref_index":32,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MSXPHTO7UQPTJ6DLYANH445JKS","json":"https://pith.science/pith/MSXPHTO7UQPTJ6DLYANH445JKS.json","graph_json":"https://pith.science/api/pith-number/MSXPHTO7UQPTJ6DLYANH445JKS/graph.json","events_json":"https://pith.science/api/pith-number/MSXPHTO7UQPTJ6DLYANH445JKS/events.json","paper":"https://pith.science/paper/MSXPHTO7"},"agent_actions":{"view_html":"https://pith.science/pith/MSXPHTO7UQPTJ6DLYANH445JKS","download_json":"https://pith.science/pith/MSXPHTO7UQPTJ6DLYANH445JKS.json","view_paper":"https://pith.science/paper/MSXPHTO7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2304.14334&json=true","fetch_graph":"https://pith.science/api/pith-number/MSXPHTO7UQPTJ6DLYANH445JKS/graph.json","fetch_events":"https://pith.science/api/pith-number/MSXPHTO7UQPTJ6DLYANH445JKS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MSXPHTO7UQPTJ6DLYANH445JKS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MSXPHTO7UQPTJ6DLYANH445JKS/action/storage_attestation","attest_author":"https://pith.science/pith/MSXPHTO7UQPTJ6DLYANH445JKS/action/author_attestation","sign_citation":"https://pith.science/pith/MSXPHTO7UQPTJ6DLYANH445JKS/action/citation_signature","submit_replication":"https://pith.science/pith/MSXPHTO7UQPTJ6DLYANH445JKS/action/replication_record"}},"created_at":"2026-07-05T06:05:03.207434+00:00","updated_at":"2026-07-05T06:05:03.207434+00:00"}