{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:L54POIEC46CU62BCPCWPEV6UOW","short_pith_number":"pith:L54POIEC","schema_version":"1.0","canonical_sha256":"5f78f72082e7854f682278acf257d47593c01fbce5bd10eba0dcd96e27e439a5","source":{"kind":"arxiv","id":"2110.12884","version":2},"attestation_state":"computed","paper":{"title":"DECAF: Generating Fair Synthetic Data Using Causally-Aware Generative Networks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Boris van Breugel, Jeroen Berrevoets, Mihaela van der Schaar, Trent Kyono","submitted_at":"2021-10-25T12:39:56Z","abstract_excerpt":"Machine learning models have been criticized for reflecting unfair biases in the training data. Instead of solving for this by introducing fair learning algorithms directly, we focus on generating fair synthetic data, such that any downstream learner is fair. Generating fair synthetic data from unfair data - while remaining truthful to the underlying data-generating process (DGP) - is non-trivial. In this paper, we introduce DECAF: a GAN-based fair synthetic data generator for tabular data. With DECAF we embed the DGP explicitly as a structural causal model in the input layers of the generator"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2110.12884","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-10-25T12:39:56Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"5fd5618936f32ce11bdebe6e67432231055d041e4d15c871c29728c154c4f2ea","abstract_canon_sha256":"f3eee941652ecf5b9ee0ed07afe2acf2d5ccbfeae35f559b26ba72b364f83f8b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:29:19.469913Z","signature_b64":"uZAIF7RkTKmtav9oz8HL1m8qbyod2tQHrxQTcDrZkppmh7tp4hrtLmuMViXCZgZOPvarppypI8kLscSgnlNDCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5f78f72082e7854f682278acf257d47593c01fbce5bd10eba0dcd96e27e439a5","last_reissued_at":"2026-07-05T03:29:19.469501Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:29:19.469501Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DECAF: Generating Fair Synthetic Data Using Causally-Aware Generative Networks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Boris van Breugel, Jeroen Berrevoets, Mihaela van der Schaar, Trent Kyono","submitted_at":"2021-10-25T12:39:56Z","abstract_excerpt":"Machine learning models have been criticized for reflecting unfair biases in the training data. Instead of solving for this by introducing fair learning algorithms directly, we focus on generating fair synthetic data, such that any downstream learner is fair. Generating fair synthetic data from unfair data - while remaining truthful to the underlying data-generating process (DGP) - is non-trivial. In this paper, we introduce DECAF: a GAN-based fair synthetic data generator for tabular data. With DECAF we embed the DGP explicitly as a structural causal model in the input layers of the generator"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2110.12884","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2110.12884/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2110.12884","created_at":"2026-07-05T03:29:19.469549+00:00"},{"alias_kind":"arxiv_version","alias_value":"2110.12884v2","created_at":"2026-07-05T03:29:19.469549+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2110.12884","created_at":"2026-07-05T03:29:19.469549+00:00"},{"alias_kind":"pith_short_12","alias_value":"L54POIEC46CU","created_at":"2026-07-05T03:29:19.469549+00:00"},{"alias_kind":"pith_short_16","alias_value":"L54POIEC46CU62BC","created_at":"2026-07-05T03:29:19.469549+00:00"},{"alias_kind":"pith_short_8","alias_value":"L54POIEC","created_at":"2026-07-05T03:29:19.469549+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2501.01785","citing_title":"Can Synthetic Data be Fair and Private? A Comparative Study of Synthetic Data Generation and Fairness Algorithms","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18878","citing_title":"Prognostic Value of Lung Ultrasound Biomarkers for Readmission Risk in Congestive Heart Failure: A Pilot Data-Driven Analysis","ref_index":256,"is_internal_anchor":false},{"citing_arxiv_id":"2509.03294","citing_title":"A Comprehensive Guide to Differential Privacy: From Theory to User Expectations","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/L54POIEC46CU62BCPCWPEV6UOW","json":"https://pith.science/pith/L54POIEC46CU62BCPCWPEV6UOW.json","graph_json":"https://pith.science/api/pith-number/L54POIEC46CU62BCPCWPEV6UOW/graph.json","events_json":"https://pith.science/api/pith-number/L54POIEC46CU62BCPCWPEV6UOW/events.json","paper":"https://pith.science/paper/L54POIEC"},"agent_actions":{"view_html":"https://pith.science/pith/L54POIEC46CU62BCPCWPEV6UOW","download_json":"https://pith.science/pith/L54POIEC46CU62BCPCWPEV6UOW.json","view_paper":"https://pith.science/paper/L54POIEC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2110.12884&json=true","fetch_graph":"https://pith.science/api/pith-number/L54POIEC46CU62BCPCWPEV6UOW/graph.json","fetch_events":"https://pith.science/api/pith-number/L54POIEC46CU62BCPCWPEV6UOW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/L54POIEC46CU62BCPCWPEV6UOW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/L54POIEC46CU62BCPCWPEV6UOW/action/storage_attestation","attest_author":"https://pith.science/pith/L54POIEC46CU62BCPCWPEV6UOW/action/author_attestation","sign_citation":"https://pith.science/pith/L54POIEC46CU62BCPCWPEV6UOW/action/citation_signature","submit_replication":"https://pith.science/pith/L54POIEC46CU62BCPCWPEV6UOW/action/replication_record"}},"created_at":"2026-07-05T03:29:19.469549+00:00","updated_at":"2026-07-05T03:29:19.469549+00:00"}