{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:ZZGOO2N66WK3GEILFQGK5T6HZS","short_pith_number":"pith:ZZGOO2N6","schema_version":"1.0","canonical_sha256":"ce4ce769bef595b3110b2c0caecfc7cc9f89096d21b5b5e52e90306016ac0f78","source":{"kind":"arxiv","id":"2310.07849","version":2},"attestation_state":"computed","paper":{"title":"Synthetic Data Generation with Large Language Models for Text Classification: Potential and Limitations","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Hangxiao Zhu, Ming Yin, Zhuoran Lu, Zhuoyan Li","submitted_at":"2023-10-11T19:51:13Z","abstract_excerpt":"The collection and curation of high-quality training data is crucial for developing text classification models with superior performance, but it is often associated with significant costs and time investment. Researchers have recently explored using large language models (LLMs) to generate synthetic datasets as an alternative approach. However, the effectiveness of the LLM-generated synthetic data in supporting model training is inconsistent across different classification tasks. To better understand factors that moderate the effectiveness of the LLM-generated synthetic data, in this study, we"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.07849","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-10-11T19:51:13Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"6a164d6f6cb55ba995af2ecac03e237284f6d7271a9abde152633ead38ec0843","abstract_canon_sha256":"c03c191566f303b695aa27ed3d05dbc551f78881d6894cb6448a4074a9d28c19"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:00:30.215594Z","signature_b64":"lAmdnfnaGTBbcc7EyI5gAwju/Jf42H6YkuCIS0tW4S0GjcDHWCJWOUxZeAqxGwNgtCwlvZt5Nv4+RvNjp2KECQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ce4ce769bef595b3110b2c0caecfc7cc9f89096d21b5b5e52e90306016ac0f78","last_reissued_at":"2026-07-05T07:00:30.215147Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:00:30.215147Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Synthetic Data Generation with Large Language Models for Text Classification: Potential and Limitations","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Hangxiao Zhu, Ming Yin, Zhuoran Lu, Zhuoyan Li","submitted_at":"2023-10-11T19:51:13Z","abstract_excerpt":"The collection and curation of high-quality training data is crucial for developing text classification models with superior performance, but it is often associated with significant costs and time investment. Researchers have recently explored using large language models (LLMs) to generate synthetic datasets as an alternative approach. However, the effectiveness of the LLM-generated synthetic data in supporting model training is inconsistent across different classification tasks. To better understand factors that moderate the effectiveness of the LLM-generated synthetic data, in this study, we"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.07849","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.07849/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.07849","created_at":"2026-07-05T07:00:30.215200+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.07849v2","created_at":"2026-07-05T07:00:30.215200+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.07849","created_at":"2026-07-05T07:00:30.215200+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZZGOO2N66WK3","created_at":"2026-07-05T07:00:30.215200+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZZGOO2N66WK3GEIL","created_at":"2026-07-05T07:00:30.215200+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZZGOO2N6","created_at":"2026-07-05T07:00:30.215200+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.04909","citing_title":"BEATS: Bootstrapping E-commerce Attribute Taxonomies for Search through Iterative Human-AI Collaboration","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2406.11354","citing_title":"Preserving Knowledge in Large Language Model with Model-Agnostic Self-Decompression","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2504.20605","citing_title":"TF1-EN-3M: Three Million Synthetic Moral Fables for Training Small, Open Language Models","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14163","citing_title":"SeaAlert: Robust Severity Classification and LLM-Based Information Extraction for Noisy Maritime Distress Communications","ref_index":31,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZZGOO2N66WK3GEILFQGK5T6HZS","json":"https://pith.science/pith/ZZGOO2N66WK3GEILFQGK5T6HZS.json","graph_json":"https://pith.science/api/pith-number/ZZGOO2N66WK3GEILFQGK5T6HZS/graph.json","events_json":"https://pith.science/api/pith-number/ZZGOO2N66WK3GEILFQGK5T6HZS/events.json","paper":"https://pith.science/paper/ZZGOO2N6"},"agent_actions":{"view_html":"https://pith.science/pith/ZZGOO2N66WK3GEILFQGK5T6HZS","download_json":"https://pith.science/pith/ZZGOO2N66WK3GEILFQGK5T6HZS.json","view_paper":"https://pith.science/paper/ZZGOO2N6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.07849&json=true","fetch_graph":"https://pith.science/api/pith-number/ZZGOO2N66WK3GEILFQGK5T6HZS/graph.json","fetch_events":"https://pith.science/api/pith-number/ZZGOO2N66WK3GEILFQGK5T6HZS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZZGOO2N66WK3GEILFQGK5T6HZS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZZGOO2N66WK3GEILFQGK5T6HZS/action/storage_attestation","attest_author":"https://pith.science/pith/ZZGOO2N66WK3GEILFQGK5T6HZS/action/author_attestation","sign_citation":"https://pith.science/pith/ZZGOO2N66WK3GEILFQGK5T6HZS/action/citation_signature","submit_replication":"https://pith.science/pith/ZZGOO2N66WK3GEILFQGK5T6HZS/action/replication_record"}},"created_at":"2026-07-05T07:00:30.215200+00:00","updated_at":"2026-07-05T07:00:30.215200+00:00"}