{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:TEHOPMEYQP3ODANC7NFAOKSXWC","short_pith_number":"pith:TEHOPMEY","schema_version":"1.0","canonical_sha256":"990ee7b09883f6e181a2fb4a072a57b08ddf4069873744fb74a460afaded9c8f","source":{"kind":"arxiv","id":"2404.14445","version":2},"attestation_state":"computed","paper":{"title":"A Multi-Faceted Evaluation Framework for Assessing Synthetic Data Generated by Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Liang Cheng, Yefeng Yuan, Yuhong Liu","submitted_at":"2024-04-20T08:08:28Z","abstract_excerpt":"The rapid advancements in generative AI and large language models (LLMs) have opened up new avenues for producing synthetic data, particularly in the realm of structured tabular formats, such as product reviews. Despite the potential benefits, concerns regarding privacy leakage have surfaced, especially when personal information is utilized in the training datasets. In addition, there is an absence of a comprehensive evaluation framework capable of quantitatively measuring the quality of the generated synthetic data and their utility for downstream tasks. In response to this gap, we introduce "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.14445","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-04-20T08:08:28Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"731e61166386662f103f68ea7fc884642e7a3c8f4f63e9f229d4a71720ac7c12","abstract_canon_sha256":"8c473b35508714366ac1e19a151341ccae532b52e3398026d6251530dd4cc58b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:42:20.767289Z","signature_b64":"+IKOvoq7u3nmxVE0ACg25CXVz/tGwqsM48k06gKcSt1liRjUE82VXb0QB+xrtNKXxju+M1ygZS5OdRky5sSGCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"990ee7b09883f6e181a2fb4a072a57b08ddf4069873744fb74a460afaded9c8f","last_reissued_at":"2026-07-05T11:42:20.766823Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:42:20.766823Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Multi-Faceted Evaluation Framework for Assessing Synthetic Data Generated by Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Liang Cheng, Yefeng Yuan, Yuhong Liu","submitted_at":"2024-04-20T08:08:28Z","abstract_excerpt":"The rapid advancements in generative AI and large language models (LLMs) have opened up new avenues for producing synthetic data, particularly in the realm of structured tabular formats, such as product reviews. Despite the potential benefits, concerns regarding privacy leakage have surfaced, especially when personal information is utilized in the training datasets. In addition, there is an absence of a comprehensive evaluation framework capable of quantitatively measuring the quality of the generated synthetic data and their utility for downstream tasks. In response to this gap, we introduce "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.14445","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.14445/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.14445","created_at":"2026-07-05T11:42:20.766877+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.14445v2","created_at":"2026-07-05T11:42:20.766877+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.14445","created_at":"2026-07-05T11:42:20.766877+00:00"},{"alias_kind":"pith_short_12","alias_value":"TEHOPMEYQP3O","created_at":"2026-07-05T11:42:20.766877+00:00"},{"alias_kind":"pith_short_16","alias_value":"TEHOPMEYQP3ODANC","created_at":"2026-07-05T11:42:20.766877+00:00"},{"alias_kind":"pith_short_8","alias_value":"TEHOPMEY","created_at":"2026-07-05T11:42:20.766877+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.14381","citing_title":"NodeSynth: Socially Aligned Synthetic Data for AI Evaluation","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2603.01444","citing_title":"Autoregressive Synthesis of Sparse and Semi-Structured Mixed-Type Data","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14381","citing_title":"NodeSynth: Socially Aligned Synthetic Data for AI Evaluation","ref_index":31,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TEHOPMEYQP3ODANC7NFAOKSXWC","json":"https://pith.science/pith/TEHOPMEYQP3ODANC7NFAOKSXWC.json","graph_json":"https://pith.science/api/pith-number/TEHOPMEYQP3ODANC7NFAOKSXWC/graph.json","events_json":"https://pith.science/api/pith-number/TEHOPMEYQP3ODANC7NFAOKSXWC/events.json","paper":"https://pith.science/paper/TEHOPMEY"},"agent_actions":{"view_html":"https://pith.science/pith/TEHOPMEYQP3ODANC7NFAOKSXWC","download_json":"https://pith.science/pith/TEHOPMEYQP3ODANC7NFAOKSXWC.json","view_paper":"https://pith.science/paper/TEHOPMEY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.14445&json=true","fetch_graph":"https://pith.science/api/pith-number/TEHOPMEYQP3ODANC7NFAOKSXWC/graph.json","fetch_events":"https://pith.science/api/pith-number/TEHOPMEYQP3ODANC7NFAOKSXWC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TEHOPMEYQP3ODANC7NFAOKSXWC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TEHOPMEYQP3ODANC7NFAOKSXWC/action/storage_attestation","attest_author":"https://pith.science/pith/TEHOPMEYQP3ODANC7NFAOKSXWC/action/author_attestation","sign_citation":"https://pith.science/pith/TEHOPMEYQP3ODANC7NFAOKSXWC/action/citation_signature","submit_replication":"https://pith.science/pith/TEHOPMEYQP3ODANC7NFAOKSXWC/action/replication_record"}},"created_at":"2026-07-05T11:42:20.766877+00:00","updated_at":"2026-07-05T11:42:20.766877+00:00"}