DPSynth generates differentially private (DP) synthetic
versions of tabular data. The model parameters are privatized; sampling
from the released model is post-processing and costs no additional
budget.
library(DPSynth)
data(adult_sample)
set.seed(1)
synth_result <- dp_synthesize(adult_sample, method = "copula",
epsilon = 2.0, delta = 1e-6,
n_synth = 300, risk_audit = FALSE)
#> Warning in regularize.values(x, y, ties, missing(ties), na.rm = na.rm):
#> collapsing to unique 'x' values
#> Warning in regularize.values(x, y, ties, missing(ties), na.rm = na.rm):
#> collapsing to unique 'x' values
#> Warning in regularize.values(x, y, ties, missing(ties), na.rm = na.rm):
#> collapsing to unique 'x' values
head(synth_result$synthetic_data)
#> age workclass education_num marital_status hours_per_week sex
#> 1 49.23322 Private 7.995714 Separated 51.57036 Female
#> 2 37.48874 Private 14.979508 Married 77.52841 Male
#> 3 40.27264 Government 8.522545 Never-married 38.00706 Female
#> 4 28.51278 Private 7.198910 Separated 39.16601 Male
#> 5 28.88234 Government 11.284476 Married 26.21400 Male
#> 6 26.58519 Private 11.636687 Married 43.14406 Male
#> capital_gain income
#> 1 3954.6 <=50K
#> 2 3954.6 <=50K
#> 3 3954.6 <=50K
#> 4 3954.6 <=50K
#> 5 3954.6 <=50K
#> 6 3954.6 <=50K
print(synth_result)
#> === Differentially Private Synthetic Data ===
#> Method: copula
#> Privacy: (epsilon = 2 , delta = 1e-06 )
#> Synthetic records: 300
#>
#> --- Utility Summary ---
#> Propensity pMSE ratio: 0.1039
#> Correlation distance: 0.1883Compare a numeric marginal:
op <- par(mfrow = c(1, 2))
hist(adult_sample$age, breaks = 20, main = "Original age",
col = "steelblue", xlab = "age")
hist(synth_result$synthetic_data$age, breaks = 20,
main = "Synthetic age", col = "indianred", xlab = "age")Other methods:
res_marg <- suppressWarnings(dp_synthesize(adult_sample,
method = "marginals",
epsilon = 2, n_synth = 300,
risk_audit = FALSE))
res_gmm <- dp_synthesize(adult_sample[, vapply(adult_sample, is.numeric,
logical(1))],
method = "gmm", epsilon = 2, n_synth = 300,
risk_audit = FALSE)
print(res_marg); print(res_gmm)
#> === Differentially Private Synthetic Data ===
#> Method: marginals
#> Privacy: (epsilon = 2 , delta = 1e-06 )
#> Synthetic records: 300
#>
#> --- Utility Summary ---
#> Propensity pMSE ratio: 0.0246
#> Correlation distance: 0.1122
#> === Differentially Private Synthetic Data ===
#> Method: gmm
#> Privacy: (epsilon = 2 , delta = 1e-06 )
#> Synthetic records: 300
#>
#> --- Utility Summary ---
#> Propensity pMSE ratio: 0.2564
#> Correlation distance: 0.3021