{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:5ZVGMJPMXAXK3XF3IHMEPR7VKB","short_pith_number":"pith:5ZVGMJPM","schema_version":"1.0","canonical_sha256":"ee6a6625ecb82eaddcbb41d847c7f5506dfa378764fa9c0815f1cccd51d1ee1c","source":{"kind":"arxiv","id":"2311.03000","version":1},"attestation_state":"computed","paper":{"title":"Strong statistical parity through fair synthetic data","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CY","stat.ML"],"primary_cat":"cs.LG","authors_text":"Ivona Krchova, Michael Platzer, Paul Tiwald","submitted_at":"2023-11-06T10:06:30Z","abstract_excerpt":"AI-generated synthetic data, in addition to protecting the privacy of original data sets, allows users and data consumers to tailor data to their needs. This paper explores the creation of synthetic data that embodies Fairness by Design, focusing on the statistical parity fairness definition. By equalizing the learned target probability distributions of the synthetic data generator across sensitive attributes, a downstream model trained on such synthetic data provides fair predictions across all thresholds, that is, strong fair predictions even when inferring from biased, original data. This f"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.03000","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-11-06T10:06:30Z","cross_cats_sorted":["cs.CY","stat.ML"],"title_canon_sha256":"8411895b73ad84d49c752677a0ad06752b71ce2dbf965b32c68dc4da168b2619","abstract_canon_sha256":"892888bfbaa970e53c888b8f768a6d8129b99fd0bba22931becb3134fc5eaa98"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:09:34.577294Z","signature_b64":"jHJ8Zij98Vfl2cjYWIwGfCuEKMvoZFcV0HqQ5L/RSobnqJ4ROIOjDeXpyaggSlm+O32aQ8YBkgks2JEuVh/ZAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ee6a6625ecb82eaddcbb41d847c7f5506dfa378764fa9c0815f1cccd51d1ee1c","last_reissued_at":"2026-07-05T07:09:34.576851Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:09:34.576851Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Strong statistical parity through fair synthetic data","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CY","stat.ML"],"primary_cat":"cs.LG","authors_text":"Ivona Krchova, Michael Platzer, Paul Tiwald","submitted_at":"2023-11-06T10:06:30Z","abstract_excerpt":"AI-generated synthetic data, in addition to protecting the privacy of original data sets, allows users and data consumers to tailor data to their needs. This paper explores the creation of synthetic data that embodies Fairness by Design, focusing on the statistical parity fairness definition. By equalizing the learned target probability distributions of the synthetic data generator across sensitive attributes, a downstream model trained on such synthetic data provides fair predictions across all thresholds, that is, strong fair predictions even when inferring from biased, original data. This f"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.03000","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.03000/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.03000","created_at":"2026-07-05T07:09:34.576901+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.03000v1","created_at":"2026-07-05T07:09:34.576901+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.03000","created_at":"2026-07-05T07:09:34.576901+00:00"},{"alias_kind":"pith_short_12","alias_value":"5ZVGMJPMXAXK","created_at":"2026-07-05T07:09:34.576901+00:00"},{"alias_kind":"pith_short_16","alias_value":"5ZVGMJPMXAXK3XF3","created_at":"2026-07-05T07:09:34.576901+00:00"},{"alias_kind":"pith_short_8","alias_value":"5ZVGMJPM","created_at":"2026-07-05T07:09:34.576901+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2501.12012","citing_title":"TabularARGN: A Flexible and Efficient Auto-Regressive Framework for Generating High-Fidelity Synthetic Data","ref_index":32,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5ZVGMJPMXAXK3XF3IHMEPR7VKB","json":"https://pith.science/pith/5ZVGMJPMXAXK3XF3IHMEPR7VKB.json","graph_json":"https://pith.science/api/pith-number/5ZVGMJPMXAXK3XF3IHMEPR7VKB/graph.json","events_json":"https://pith.science/api/pith-number/5ZVGMJPMXAXK3XF3IHMEPR7VKB/events.json","paper":"https://pith.science/paper/5ZVGMJPM"},"agent_actions":{"view_html":"https://pith.science/pith/5ZVGMJPMXAXK3XF3IHMEPR7VKB","download_json":"https://pith.science/pith/5ZVGMJPMXAXK3XF3IHMEPR7VKB.json","view_paper":"https://pith.science/paper/5ZVGMJPM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.03000&json=true","fetch_graph":"https://pith.science/api/pith-number/5ZVGMJPMXAXK3XF3IHMEPR7VKB/graph.json","fetch_events":"https://pith.science/api/pith-number/5ZVGMJPMXAXK3XF3IHMEPR7VKB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5ZVGMJPMXAXK3XF3IHMEPR7VKB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5ZVGMJPMXAXK3XF3IHMEPR7VKB/action/storage_attestation","attest_author":"https://pith.science/pith/5ZVGMJPMXAXK3XF3IHMEPR7VKB/action/author_attestation","sign_citation":"https://pith.science/pith/5ZVGMJPMXAXK3XF3IHMEPR7VKB/action/citation_signature","submit_replication":"https://pith.science/pith/5ZVGMJPMXAXK3XF3IHMEPR7VKB/action/replication_record"}},"created_at":"2026-07-05T07:09:34.576901+00:00","updated_at":"2026-07-05T07:09:34.576901+00:00"}