{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:S2J2UMIPISLNWAB2VQNZYCN3FE","short_pith_number":"pith:S2J2UMIP","schema_version":"1.0","canonical_sha256":"9693aa310f4496db003aac1b9c09bb290564ef63bf588f32589a4b6391177247","source":{"kind":"arxiv","id":"2305.10015","version":4},"attestation_state":"computed","paper":{"title":"Utility Theory of Synthetic Data Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Guang Cheng, Shirong Xu, Will Wei Sun","submitted_at":"2023-05-17T07:49:16Z","abstract_excerpt":"Synthetic data algorithms are widely employed in industries to generate artificial data for downstream learning tasks. While existing research primarily focuses on empirically evaluating utility of synthetic data, its theoretical understanding is largely lacking. This paper bridges the practice-theory gap by establishing relevant utility theory in a statistical learning framework. It considers two utility metrics: generalization and ranking of models trained on synthetic data. The former is defined as the generalization difference between models trained on synthetic and on real data. By derivi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.10015","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"stat.ML","submitted_at":"2023-05-17T07:49:16Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"0ec44fe344a69a6937f4c036e69b1cb122404b6a8f75c2f3fbf18b0b98600154","abstract_canon_sha256":"84dcb5570795f648a1d0379d39fe1bc700e50970a9992b8ab3eecd4a21e91eb9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:43:33.473648Z","signature_b64":"r2OPeRh8fICvKnzsbwsI0wHKB0QKAj7NsaNkn2GsyvWO6ZdnwNo2SU2MNghPH1lFPQYrfPfCJEip47ntUbixBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9693aa310f4496db003aac1b9c09bb290564ef63bf588f32589a4b6391177247","last_reissued_at":"2026-07-05T10:43:33.473198Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:43:33.473198Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Utility Theory of Synthetic Data Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Guang Cheng, Shirong Xu, Will Wei Sun","submitted_at":"2023-05-17T07:49:16Z","abstract_excerpt":"Synthetic data algorithms are widely employed in industries to generate artificial data for downstream learning tasks. While existing research primarily focuses on empirically evaluating utility of synthetic data, its theoretical understanding is largely lacking. This paper bridges the practice-theory gap by establishing relevant utility theory in a statistical learning framework. It considers two utility metrics: generalization and ranking of models trained on synthetic data. The former is defined as the generalization difference between models trained on synthetic and on real data. By derivi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.10015","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.10015/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.10015","created_at":"2026-07-05T10:43:33.473259+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.10015v4","created_at":"2026-07-05T10:43:33.473259+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.10015","created_at":"2026-07-05T10:43:33.473259+00:00"},{"alias_kind":"pith_short_12","alias_value":"S2J2UMIPISLN","created_at":"2026-07-05T10:43:33.473259+00:00"},{"alias_kind":"pith_short_16","alias_value":"S2J2UMIPISLNWAB2","created_at":"2026-07-05T10:43:33.473259+00:00"},{"alias_kind":"pith_short_8","alias_value":"S2J2UMIP","created_at":"2026-07-05T10:43:33.473259+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.10088","citing_title":"Towards High Supervised Learning Utility Training Data Generation: Data Pruning and Column Reordering","ref_index":73,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/S2J2UMIPISLNWAB2VQNZYCN3FE","json":"https://pith.science/pith/S2J2UMIPISLNWAB2VQNZYCN3FE.json","graph_json":"https://pith.science/api/pith-number/S2J2UMIPISLNWAB2VQNZYCN3FE/graph.json","events_json":"https://pith.science/api/pith-number/S2J2UMIPISLNWAB2VQNZYCN3FE/events.json","paper":"https://pith.science/paper/S2J2UMIP"},"agent_actions":{"view_html":"https://pith.science/pith/S2J2UMIPISLNWAB2VQNZYCN3FE","download_json":"https://pith.science/pith/S2J2UMIPISLNWAB2VQNZYCN3FE.json","view_paper":"https://pith.science/paper/S2J2UMIP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.10015&json=true","fetch_graph":"https://pith.science/api/pith-number/S2J2UMIPISLNWAB2VQNZYCN3FE/graph.json","fetch_events":"https://pith.science/api/pith-number/S2J2UMIPISLNWAB2VQNZYCN3FE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/S2J2UMIPISLNWAB2VQNZYCN3FE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/S2J2UMIPISLNWAB2VQNZYCN3FE/action/storage_attestation","attest_author":"https://pith.science/pith/S2J2UMIPISLNWAB2VQNZYCN3FE/action/author_attestation","sign_citation":"https://pith.science/pith/S2J2UMIPISLNWAB2VQNZYCN3FE/action/citation_signature","submit_replication":"https://pith.science/pith/S2J2UMIPISLNWAB2VQNZYCN3FE/action/replication_record"}},"created_at":"2026-07-05T10:43:33.473259+00:00","updated_at":"2026-07-05T10:43:33.473259+00:00"}