Getting Started with DPSynth

Getting Started with DPSynth

DPSynth generates differentially private (DP) synthetic versions of tabular data. The model parameters are privatized; sampling from the released model is post-processing and costs no additional budget.

library(DPSynth)

data(adult_sample)
set.seed(1)
synth_result <- dp_synthesize(adult_sample, method = "copula",
                              epsilon = 2.0, delta = 1e-6,
                              n_synth = 300, risk_audit = FALSE)
#> Warning in regularize.values(x, y, ties, missing(ties), na.rm = na.rm):
#> collapsing to unique 'x' values
#> Warning in regularize.values(x, y, ties, missing(ties), na.rm = na.rm):
#> collapsing to unique 'x' values
#> Warning in regularize.values(x, y, ties, missing(ties), na.rm = na.rm):
#> collapsing to unique 'x' values
head(synth_result$synthetic_data)
#>        age  workclass education_num marital_status hours_per_week    sex
#> 1 49.23322    Private      7.995714      Separated       51.57036 Female
#> 2 37.48874    Private     14.979508        Married       77.52841   Male
#> 3 40.27264 Government      8.522545  Never-married       38.00706 Female
#> 4 28.51278    Private      7.198910      Separated       39.16601   Male
#> 5 28.88234 Government     11.284476        Married       26.21400   Male
#> 6 26.58519    Private     11.636687        Married       43.14406   Male
#>   capital_gain income
#> 1       3954.6  <=50K
#> 2       3954.6  <=50K
#> 3       3954.6  <=50K
#> 4       3954.6  <=50K
#> 5       3954.6  <=50K
#> 6       3954.6  <=50K
print(synth_result)
#> === Differentially Private Synthetic Data ===
#> Method: copula 
#> Privacy: (epsilon = 2 , delta = 1e-06 )
#> Synthetic records: 300 
#> 
#> --- Utility Summary ---
#> Propensity pMSE ratio: 0.1039 
#> Correlation distance: 0.1883

Compare a numeric marginal:

op <- par(mfrow = c(1, 2))
hist(adult_sample$age, breaks = 20, main = "Original age",
     col = "steelblue", xlab = "age")
hist(synth_result$synthetic_data$age, breaks = 20,
     main = "Synthetic age", col = "indianred", xlab = "age")

par(op)

Other methods:

res_marg <- suppressWarnings(dp_synthesize(adult_sample,
                                           method = "marginals",
                                           epsilon = 2, n_synth = 300,
                                           risk_audit = FALSE))
res_gmm <- dp_synthesize(adult_sample[, vapply(adult_sample, is.numeric,
                                               logical(1))],
                         method = "gmm", epsilon = 2, n_synth = 300,
                         risk_audit = FALSE)
print(res_marg); print(res_gmm)
#> === Differentially Private Synthetic Data ===
#> Method: marginals 
#> Privacy: (epsilon = 2 , delta = 1e-06 )
#> Synthetic records: 300 
#> 
#> --- Utility Summary ---
#> Propensity pMSE ratio: 0.0246 
#> Correlation distance: 0.1122
#> === Differentially Private Synthetic Data ===
#> Method: gmm 
#> Privacy: (epsilon = 2 , delta = 1e-06 )
#> Synthetic records: 300 
#> 
#> --- Utility Summary ---
#> Propensity pMSE ratio: 0.2564 
#> Correlation distance: 0.3021