{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:SX7TSS2CGVM6SCPGCWV3YER372","short_pith_number":"pith:SX7TSS2C","schema_version":"1.0","canonical_sha256":"95ff394b423559e909e615abbc123bfe8979e9b95a0be31a54f424ffdad29907","source":{"kind":"arxiv","id":"2504.20687","version":1},"attestation_state":"computed","paper":{"title":"What's Wrong with Your Synthetic Tabular Data? Using Explainable AI to Evaluate Generative Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Jan Kapar, Martin Jullum, Niklas Koenen","submitted_at":"2025-04-29T12:10:52Z","abstract_excerpt":"Evaluating synthetic tabular data is challenging, since they can differ from the real data in so many ways. There exist numerous metrics of synthetic data quality, ranging from statistical distances to predictive performance, often providing conflicting results. Moreover, they fail to explain or pinpoint the specific weaknesses in the synthetic data. To address this, we apply explainable AI (XAI) techniques to a binary detection classifier trained to distinguish real from synthetic data. While the classifier identifies distributional differences, XAI concepts such as feature importance and fea"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.20687","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-04-29T12:10:52Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"0f1e0c728c3b3d5e7988ad07911d5242bacaf1f1288aa09015f56864109020a5","abstract_canon_sha256":"a0c2e2cff9c0d30a08898c89f8c89ef8640014e5871a34d3bf3c16f704e5b1da"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:55:47.153225Z","signature_b64":"mgmBAEgRN8eLNcgjVomhYqV8y6iIJhwYgvVxw934AIOAX4q9SGbhuepSN3JYAKLE9MGE+bHX/BZpnHxeFLRuAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"95ff394b423559e909e615abbc123bfe8979e9b95a0be31a54f424ffdad29907","last_reissued_at":"2026-07-05T10:55:47.152751Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:55:47.152751Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"What's Wrong with Your Synthetic Tabular Data? Using Explainable AI to Evaluate Generative Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Jan Kapar, Martin Jullum, Niklas Koenen","submitted_at":"2025-04-29T12:10:52Z","abstract_excerpt":"Evaluating synthetic tabular data is challenging, since they can differ from the real data in so many ways. There exist numerous metrics of synthetic data quality, ranging from statistical distances to predictive performance, often providing conflicting results. Moreover, they fail to explain or pinpoint the specific weaknesses in the synthetic data. To address this, we apply explainable AI (XAI) techniques to a binary detection classifier trained to distinguish real from synthetic data. While the classifier identifies distributional differences, XAI concepts such as feature importance and fea"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.20687","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.20687/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.20687","created_at":"2026-07-05T10:55:47.152814+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.20687v1","created_at":"2026-07-05T10:55:47.152814+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.20687","created_at":"2026-07-05T10:55:47.152814+00:00"},{"alias_kind":"pith_short_12","alias_value":"SX7TSS2CGVM6","created_at":"2026-07-05T10:55:47.152814+00:00"},{"alias_kind":"pith_short_16","alias_value":"SX7TSS2CGVM6SCPG","created_at":"2026-07-05T10:55:47.152814+00:00"},{"alias_kind":"pith_short_8","alias_value":"SX7TSS2C","created_at":"2026-07-05T10:55:47.152814+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SX7TSS2CGVM6SCPGCWV3YER372","json":"https://pith.science/pith/SX7TSS2CGVM6SCPGCWV3YER372.json","graph_json":"https://pith.science/api/pith-number/SX7TSS2CGVM6SCPGCWV3YER372/graph.json","events_json":"https://pith.science/api/pith-number/SX7TSS2CGVM6SCPGCWV3YER372/events.json","paper":"https://pith.science/paper/SX7TSS2C"},"agent_actions":{"view_html":"https://pith.science/pith/SX7TSS2CGVM6SCPGCWV3YER372","download_json":"https://pith.science/pith/SX7TSS2CGVM6SCPGCWV3YER372.json","view_paper":"https://pith.science/paper/SX7TSS2C","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.20687&json=true","fetch_graph":"https://pith.science/api/pith-number/SX7TSS2CGVM6SCPGCWV3YER372/graph.json","fetch_events":"https://pith.science/api/pith-number/SX7TSS2CGVM6SCPGCWV3YER372/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SX7TSS2CGVM6SCPGCWV3YER372/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SX7TSS2CGVM6SCPGCWV3YER372/action/storage_attestation","attest_author":"https://pith.science/pith/SX7TSS2CGVM6SCPGCWV3YER372/action/author_attestation","sign_citation":"https://pith.science/pith/SX7TSS2CGVM6SCPGCWV3YER372/action/citation_signature","submit_replication":"https://pith.science/pith/SX7TSS2CGVM6SCPGCWV3YER372/action/replication_record"}},"created_at":"2026-07-05T10:55:47.152814+00:00","updated_at":"2026-07-05T10:55:47.152814+00:00"}