Health records mix continuous lab values, categorical diagnoses and outcomes. This vignette synthesizes such data and audits disclosure risk.
library(DPSynth)
set.seed(3)
n <- 400
ehr <- data.frame(
age = as.integer(pmin(100, pmax(18, round(rnorm(n, 60, 15))))),
labs = rnorm(n, 5, 1.5),
diagnosis = factor(sample(c("I10", "E11", "J44", "N18"), n,
replace = TRUE)),
readmitted = factor(sample(c("no", "<30d", ">30d"), n,
prob = c(.6, .15, .25), replace = TRUE))
)
res <- dp_synthesize(ehr, method = "copula", epsilon = 2, n_synth = 400)
#> Warning in regularize.values(x, y, ties, missing(ties), na.rm = na.rm):
#> collapsing to unique 'x' values
head(res$synthetic_data)
#> age labs diagnosis readmitted
#> 1 58.34062 3.923699 E11 >30d
#> 2 62.10151 5.390274 N18 >30d
#> 3 53.97709 5.221595 N18 no
#> 4 39.62570 4.437263 I10 no
#> 5 66.53093 6.856470 J44 no
#> 6 56.85954 6.879139 J44 no
print(res)
#> === Differentially Private Synthetic Data ===
#> Method: copula
#> Privacy: (epsilon = 2 , delta = 1e-06 )
#> Synthetic records: 400
#>
#> --- Utility Summary ---
#> Propensity pMSE ratio: 0.0068
#> Correlation distance: 0.0512
#>
#> --- Risk Summary ---
#> Membership risk score: 0.05
# Empirical disclosure risk beyond the theoretical epsilon guarantee
res$risk$membership$risk_score
#> [1] 0.05
res$risk$attribute$mean_disclosure_error
#> [1] NADownstream validation – predicting readmission from synthetic training data and testing on real held-out records: