Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
12 changes: 9 additions & 3 deletions dpsynth/_calibration.py
Original file line number Diff line number Diff line change
Expand Up @@ -30,6 +30,7 @@ def calibrate(
*,
epsilon: float,
delta: float,
delta_split: float = 0.5,
poisson_sampling_prob: float = 1.0,
max_records_per_user: int = 1,
accountant_fn: Callable[[], dp_accounting.PrivacyAccountant] | None = None,
Expand All @@ -46,6 +47,9 @@ def calibrate(
domain: Optional domain specification, forwarded to ``config.configure()``.
epsilon: Target epsilon for (epsilon, delta)-DP.
delta: Target delta for (epsilon, delta)-DP.
delta_split: Fraction of ``delta`` passed to ``config.configure()`` for
sub-mechanisms that consume approximate DP budget directly (e.g. open-set
partition selection). Defaults to 0.5.
poisson_sampling_prob: If specified, calibrate the mechanism assuming the
input data is subsampled with the given probability. The actual sampling
is **NOT** handled internally by the calibrated mechanism.
Expand All @@ -61,17 +65,19 @@ def calibrate(
A calibrated, runnable mechanism.

Raises:
ValueError: If epsilon is not positive.
ValueError: If epsilon is not positive or delta_split is not in (0, 1).
UnsupportedEventError: If no accountant supports the mechanism.
"""
if epsilon <= 0:
raise ValueError(f'Target epsilon must be positive, got {epsilon}.')
if not 0 < delta_split < 1:
raise ValueError(f'delta_split must be in (0, 1), got {delta_split}.')

def make_event_fn(rho: float) -> dp_accounting.DpEvent:
base = config.configure(
domain,
budget=rho,
delta=delta,
delta=delta * delta_split,
max_records_per_user=max_records_per_user,
).dp_event
sampled = dp_accounting.PoissonSampledDpEvent(poisson_sampling_prob, base)
Expand Down Expand Up @@ -116,6 +122,6 @@ def make_event_fn(rho: float) -> dp_accounting.DpEvent:
return config.configure(
domain,
budget=optimal_rho,
delta=delta,
delta=delta * delta_split,
max_records_per_user=max_records_per_user,
)
2 changes: 2 additions & 0 deletions dpsynth/api.py
Original file line number Diff line number Diff line change
Expand Up @@ -173,6 +173,7 @@ def calibrate(
*,
epsilon: float,
delta: float,
delta_split: float = 0.5,
poisson_sampling_prob: float = 1.0,
max_records_per_user: int = 1,
accountant_fn: (
Expand All @@ -191,6 +192,7 @@ def calibrate(
domain,
epsilon=epsilon,
delta=delta,
delta_split=delta_split,
poisson_sampling_prob=poisson_sampling_prob,
max_records_per_user=max_records_per_user,
accountant_fn=accountant_fn,
Expand Down
30 changes: 10 additions & 20 deletions dpsynth/data_generation_v3.py
Original file line number Diff line number Diff line change
Expand Up @@ -419,10 +419,9 @@ class TabularConfig(api.MechanismConfig):
use_jax_for_generation: bool = False

def _compute_per_col_deltas(self, domains, delta):
# Split delta across open-set columns, analogous to splitting zcdp_rho.
# Under calibrate(), any delta not consumed here is automatically
# available for the zCDP-to-(epsilon, delta) conversion, so this
# simple additive split is tight.
# Split delta equally across open-set columns. Under calibrate(), any
# delta not consumed here is automatically available for the
# zCDP-to-(epsilon, delta) conversion, so this additive split is tight.
num_open_set = sum(
isinstance(attr, domain.OpenSetCategoricalAttribute)
for attr in domains.values()
Expand All @@ -433,12 +432,10 @@ def _compute_per_col_deltas(self, domains, delta):
' present. It is used for Gaussian partition selection.'
)

thresholding_delta = self.init_budget_fraction * delta

per_col_deltas = {}
for col in domains:
if isinstance(domains[col], domain.OpenSetCategoricalAttribute):
per_col_deltas[col] = thresholding_delta / num_open_set
per_col_deltas[col] = delta / num_open_set
else:
per_col_deltas[col] = 0.0
return per_col_deltas
Expand All @@ -453,27 +450,20 @@ def configure(
) -> TabularMechanism:
"""Returns a calibrated mechanism configured with the given privacy budget.

Splits the budget additively, just as it does for ``budget``:
Splits the budget additively:

- ``init_budget_fraction`` of ``budget`` goes to per-column initializers
(split evenly, including a total-count mechanism); the remainder goes to
the discrete mechanism.
- ``init_budget_fraction`` of ``delta`` is reserved for open-set partition
selection (split evenly across open-set columns); the remaining delta is
unused by pure-zCDP sub-mechanisms.

When ``calibrate(epsilon, delta)`` is called, the base class binary search
passes the guarantee delta here. Because the thresholding delta is honestly
reported in the composite ``dp_event``, the binary search automatically
ensures the overall (epsilon, delta) guarantee is tight.
- ``delta`` is split evenly across open-set columns for partition selection
(pure-zCDP sub-mechanisms do not consume ``delta``).

Args:
schema: Dataset schema or mapping from column names to attribute domain.
budget: The privacy budget.
delta: Overall approximate DP delta for the mechanism. A fraction
(``init_budget_fraction``) is allocated to partition selection for
open-set columns. Must be positive when open-set categorical attributes
are present.
delta: Approximate DP delta allocated to partition selection for open-set
columns (split evenly across open-set columns). Must be positive when
open-set categorical attributes are present.
max_records_per_user: Assumed upper bound on the number of records a
single user contributes. Values greater than 1 scale the added noise
(and mechanism sensitivity) to provide user-level rather than
Expand Down
14 changes: 3 additions & 11 deletions dpsynth/relational/synthesizer.py
Original file line number Diff line number Diff line change
Expand Up @@ -426,7 +426,6 @@ def _create_table_initializers(
def _compute_table_col_deltas(
domains: Mapping[str, domain.Schema],
delta: float,
init_budget_fraction: float,
) -> dict[str, dict[str, float]]:
"""Splits thresholding delta additively across open-set columns in all tables.

Expand All @@ -437,7 +436,6 @@ def _compute_table_col_deltas(
Args:
domains: Mapping from table names to per-column AttributeType schemas.
delta: Total DP delta for partition selection thresholding.
init_budget_fraction: Fraction of delta allocated to column initialization.

Returns:
A nested mapping from table name and column name to its allocated delta.
Expand All @@ -448,8 +446,7 @@ def _compute_table_col_deltas(
Formal Guarantees:
- Only open-set categorical attributes consume delta
- Categorical and numerical attributes operate under pure zCDP (delta = 0.0)
- Sum of per-column deltas across all tables equals init_budget_fraction *
delta.
- Sum of per-column deltas across all tables equals delta.
- Invariance: If no open-set columns exist, all per-column deltas are 0.0.
"""
num_open_set = 0
Expand All @@ -462,8 +459,7 @@ def _compute_table_col_deltas(
'delta must be positive when open-set categorical attributes are'
' present. It is used for Gaussian partition selection.'
)
thresholding_delta = init_budget_fraction * delta
per_col_delta = thresholding_delta / num_open_set if num_open_set > 0 else 0.0
per_col_delta = delta / num_open_set if num_open_set > 0 else 0.0
return {
table: {
col: (
Expand Down Expand Up @@ -1259,11 +1255,7 @@ def configure(
hierarchy, max_records_per_user=max_records_per_user
)

per_col_deltas = _compute_table_col_deltas(
domains,
delta=delta,
init_budget_fraction=self.init_budget_fraction,
)
per_col_deltas = _compute_table_col_deltas(domains, delta=delta)
inits = (
self.initializers
if self.initializers is not None
Expand Down
23 changes: 23 additions & 0 deletions tests/data_generation_v3_test.py
Original file line number Diff line number Diff line change
Expand Up @@ -239,6 +239,29 @@ def test_calibrate_domain_positional_only(self):
with self.assertRaises(TypeError):
dpsynth.calibrate(config, domain=domains, epsilon=1.0, delta=1e-5) # pyrefly: ignore[unexpected-keyword]

def test_calibrate_delta_split(self):
domains = {
'A': domain.OpenSetCategoricalAttribute(),
'B': domain.OpenSetCategoricalAttribute(),
'C': domain.CategoricalAttribute(possible_values=['x', 'y']),
}
config = TabularConfig()
calibrated = dpsynth.calibrate(config, domains, epsilon=1.0, delta=1e-5)
self.assertAlmostEqual(calibrated.initializers['A'].delta, 2.5e-6)
self.assertAlmostEqual(calibrated.initializers['B'].delta, 2.5e-6)

custom = dpsynth.calibrate(
config, domains, epsilon=1.0, delta=1e-5, delta_split=0.2
)
self.assertAlmostEqual(custom.initializers['A'].delta, 1e-6)
self.assertAlmostEqual(custom.initializers['B'].delta, 1e-6)

for bad_split in (0.0, 1.0, -0.1, 1.5):
with self.assertRaisesRegex(ValueError, 'delta_split must be in'):
dpsynth.calibrate(
config, domains, epsilon=1.0, delta=1e-5, delta_split=bad_split
)

@parameterized.product(
sentinel=[np.nan, None],
clip_to_range=[True, False],
Expand Down
7 changes: 7 additions & 0 deletions tests/local_mode/initialization_test.py
Original file line number Diff line number Diff line change
Expand Up @@ -288,6 +288,13 @@ def test_dp_event(self):
)
self.assertEqual(event.events[1].delta, 1e-5)

def test_calibrate(self):
attr = domain.OpenSetCategoricalAttribute(default_value='<OOD>')
config = initialization.OpenSetInitializerConfig()
calibrated = config.calibrate(attr, epsilon=1.0, delta=1e-5)
self.assertAlmostEqual(calibrated.delta, 5e-6)
self.assertGreater(calibrated.sigma, 0.0)

def test_call_noiseless(self):
attr = domain.OpenSetCategoricalAttribute(default_value='<OOD>')
rng = np.random.default_rng(42)
Expand Down
19 changes: 6 additions & 13 deletions tests/relational/synthesizer_test.py
Original file line number Diff line number Diff line change
Expand Up @@ -97,14 +97,11 @@ def test_compute_table_col_deltas_with_open_set(self):
'gender': domain.CategoricalAttribute(possible_values=['M', 'F']),
},
}
deltas = synthesizer._compute_table_col_deltas(
domains, delta=1e-4, init_budget_fraction=0.2
)
# Total thresholding delta = 0.2 * 1e-4 = 2e-5.
# 2 open-set columns -> each gets 1e-5.
self.assertAlmostEqual(deltas['Household']['tags'], 1e-5)
deltas = synthesizer._compute_table_col_deltas(domains, delta=1e-4)
# 2 open-set columns -> each gets 5e-5.
self.assertAlmostEqual(deltas['Household']['tags'], 5e-5)
self.assertEqual(deltas['Household']['income'], 0.0)
self.assertAlmostEqual(deltas['Person']['hobbies'], 1e-5)
self.assertAlmostEqual(deltas['Person']['hobbies'], 5e-5)
self.assertEqual(deltas['Person']['gender'], 0.0)

def test_compute_table_col_deltas_no_open_set(self):
Expand All @@ -114,9 +111,7 @@ def test_compute_table_col_deltas_no_open_set(self):
'region': domain.CategoricalAttribute(possible_values=['U', 'R']),
},
}
deltas = synthesizer._compute_table_col_deltas(
domains, delta=0.0, init_budget_fraction=0.1
)
deltas = synthesizer._compute_table_col_deltas(domains, delta=0.0)
self.assertEqual(deltas['Household']['income'], 0.0)
self.assertEqual(deltas['Household']['region'], 0.0)

Expand All @@ -127,9 +122,7 @@ def test_compute_table_col_deltas_missing_delta_raises(self):
},
}
with self.assertRaisesRegex(ValueError, 'delta must be positive'):
synthesizer._compute_table_col_deltas(
domains, delta=0.0, init_budget_fraction=0.1
)
synthesizer._compute_table_col_deltas(domains, delta=0.0)

def test_dp_event_composition(self):
# Setup calibrated initializers for 2 tables.
Expand Down
Loading