{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:4NECOP2KGPLWWN6JSKHDEIGAIV","short_pith_number":"pith:4NECOP2K","schema_version":"1.0","canonical_sha256":"e348273f4a33d76b37c9928e3220c0455acf50b4f8c53adeda30d245b77aff06","source":{"kind":"arxiv","id":"2409.19759","version":3},"attestation_state":"computed","paper":{"title":"Balancing Cost and Effectiveness of Synthetic Data Generation Strategies for LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Apaar Shanker, George Pu, John Heyer, Parth Suresh, Penn Jenks, Sam Denton, Yung-Chieh Chan","submitted_at":"2024-09-29T20:14:50Z","abstract_excerpt":"As large language models (LLMs) are applied to more use cases, creating high quality, task-specific datasets for fine-tuning becomes a bottleneck for model improvement. Using high quality human data has been the most common approach to unlock model performance, but is prohibitively expensive in many scenarios. Several alternative methods have also emerged, such as generating synthetic or hybrid data, but the effectiveness of these approaches remain unclear, especially in resource-constrained scenarios and tasks that are not easily verified. To investigate this, we group various synthetic data "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.19759","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-09-29T20:14:50Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"3ad9f5585b694458e9f111dd77ef3f61cb98aa104d9ecc3af468ca43a7279e49","abstract_canon_sha256":"f4919e0f71465cdf10a042d885576d51e31c47c3da392f1bac17b9aa549d3ca8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:28:13.959090Z","signature_b64":"oUBFsknJFXzmRENkIT/tvNfUJML8GSiuYJnwwPbArup9u2W23vx1CWLZ648e8+HW1qDmGILlzoMppCwiTWrRBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e348273f4a33d76b37c9928e3220c0455acf50b4f8c53adeda30d245b77aff06","last_reissued_at":"2026-07-05T09:28:13.958650Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:28:13.958650Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Balancing Cost and Effectiveness of Synthetic Data Generation Strategies for LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Apaar Shanker, George Pu, John Heyer, Parth Suresh, Penn Jenks, Sam Denton, Yung-Chieh Chan","submitted_at":"2024-09-29T20:14:50Z","abstract_excerpt":"As large language models (LLMs) are applied to more use cases, creating high quality, task-specific datasets for fine-tuning becomes a bottleneck for model improvement. Using high quality human data has been the most common approach to unlock model performance, but is prohibitively expensive in many scenarios. Several alternative methods have also emerged, such as generating synthetic or hybrid data, but the effectiveness of these approaches remain unclear, especially in resource-constrained scenarios and tasks that are not easily verified. To investigate this, we group various synthetic data "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.19759","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.19759/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.19759","created_at":"2026-07-05T09:28:13.958706+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.19759v3","created_at":"2026-07-05T09:28:13.958706+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.19759","created_at":"2026-07-05T09:28:13.958706+00:00"},{"alias_kind":"pith_short_12","alias_value":"4NECOP2KGPLW","created_at":"2026-07-05T09:28:13.958706+00:00"},{"alias_kind":"pith_short_16","alias_value":"4NECOP2KGPLWWN6J","created_at":"2026-07-05T09:28:13.958706+00:00"},{"alias_kind":"pith_short_8","alias_value":"4NECOP2K","created_at":"2026-07-05T09:28:13.958706+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.24977","citing_title":"A Survey on LLM-based Conversational User Simulation","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4NECOP2KGPLWWN6JSKHDEIGAIV","json":"https://pith.science/pith/4NECOP2KGPLWWN6JSKHDEIGAIV.json","graph_json":"https://pith.science/api/pith-number/4NECOP2KGPLWWN6JSKHDEIGAIV/graph.json","events_json":"https://pith.science/api/pith-number/4NECOP2KGPLWWN6JSKHDEIGAIV/events.json","paper":"https://pith.science/paper/4NECOP2K"},"agent_actions":{"view_html":"https://pith.science/pith/4NECOP2KGPLWWN6JSKHDEIGAIV","download_json":"https://pith.science/pith/4NECOP2KGPLWWN6JSKHDEIGAIV.json","view_paper":"https://pith.science/paper/4NECOP2K","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.19759&json=true","fetch_graph":"https://pith.science/api/pith-number/4NECOP2KGPLWWN6JSKHDEIGAIV/graph.json","fetch_events":"https://pith.science/api/pith-number/4NECOP2KGPLWWN6JSKHDEIGAIV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4NECOP2KGPLWWN6JSKHDEIGAIV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4NECOP2KGPLWWN6JSKHDEIGAIV/action/storage_attestation","attest_author":"https://pith.science/pith/4NECOP2KGPLWWN6JSKHDEIGAIV/action/author_attestation","sign_citation":"https://pith.science/pith/4NECOP2KGPLWWN6JSKHDEIGAIV/action/citation_signature","submit_replication":"https://pith.science/pith/4NECOP2KGPLWWN6JSKHDEIGAIV/action/replication_record"}},"created_at":"2026-07-05T09:28:13.958706+00:00","updated_at":"2026-07-05T09:28:13.958706+00:00"}