diff --git a/dpsynth/_calibration.py b/dpsynth/_calibration.py index 1ecdaf0f..e5d5f21b 100644 --- a/dpsynth/_calibration.py +++ b/dpsynth/_calibration.py @@ -30,6 +30,7 @@ def calibrate( *, epsilon: float, delta: float, + delta_split: float = 0.5, poisson_sampling_prob: float = 1.0, max_records_per_user: int = 1, accountant_fn: Callable[[], dp_accounting.PrivacyAccountant] | None = None, @@ -46,6 +47,9 @@ def calibrate( domain: Optional domain specification, forwarded to ``config.configure()``. epsilon: Target epsilon for (epsilon, delta)-DP. delta: Target delta for (epsilon, delta)-DP. + delta_split: Fraction of ``delta`` passed to ``config.configure()`` for + sub-mechanisms that consume approximate DP budget directly (e.g. open-set + partition selection). Defaults to 0.5. poisson_sampling_prob: If specified, calibrate the mechanism assuming the input data is subsampled with the given probability. The actual sampling is **NOT** handled internally by the calibrated mechanism. @@ -61,17 +65,19 @@ def calibrate( A calibrated, runnable mechanism. Raises: - ValueError: If epsilon is not positive. + ValueError: If epsilon is not positive or delta_split is not in (0, 1). UnsupportedEventError: If no accountant supports the mechanism. """ if epsilon <= 0: raise ValueError(f'Target epsilon must be positive, got {epsilon}.') + if not 0 < delta_split < 1: + raise ValueError(f'delta_split must be in (0, 1), got {delta_split}.') def make_event_fn(rho: float) -> dp_accounting.DpEvent: base = config.configure( domain, budget=rho, - delta=delta, + delta=delta * delta_split, max_records_per_user=max_records_per_user, ).dp_event sampled = dp_accounting.PoissonSampledDpEvent(poisson_sampling_prob, base) @@ -116,6 +122,6 @@ def make_event_fn(rho: float) -> dp_accounting.DpEvent: return config.configure( domain, budget=optimal_rho, - delta=delta, + delta=delta * delta_split, max_records_per_user=max_records_per_user, ) diff --git a/dpsynth/api.py b/dpsynth/api.py index 4340ddae..32485efb 100644 --- a/dpsynth/api.py +++ b/dpsynth/api.py @@ -173,6 +173,7 @@ def calibrate( *, epsilon: float, delta: float, + delta_split: float = 0.5, poisson_sampling_prob: float = 1.0, max_records_per_user: int = 1, accountant_fn: ( @@ -191,6 +192,7 @@ def calibrate( domain, epsilon=epsilon, delta=delta, + delta_split=delta_split, poisson_sampling_prob=poisson_sampling_prob, max_records_per_user=max_records_per_user, accountant_fn=accountant_fn, diff --git a/dpsynth/data_generation_v3.py b/dpsynth/data_generation_v3.py index e093d515..fdfbf2c3 100644 --- a/dpsynth/data_generation_v3.py +++ b/dpsynth/data_generation_v3.py @@ -419,10 +419,9 @@ class TabularConfig(api.MechanismConfig): use_jax_for_generation: bool = False def _compute_per_col_deltas(self, domains, delta): - # Split delta across open-set columns, analogous to splitting zcdp_rho. - # Under calibrate(), any delta not consumed here is automatically - # available for the zCDP-to-(epsilon, delta) conversion, so this - # simple additive split is tight. + # Split delta equally across open-set columns. Under calibrate(), any + # delta not consumed here is automatically available for the + # zCDP-to-(epsilon, delta) conversion, so this additive split is tight. num_open_set = sum( isinstance(attr, domain.OpenSetCategoricalAttribute) for attr in domains.values() @@ -433,12 +432,10 @@ def _compute_per_col_deltas(self, domains, delta): ' present. It is used for Gaussian partition selection.' ) - thresholding_delta = self.init_budget_fraction * delta - per_col_deltas = {} for col in domains: if isinstance(domains[col], domain.OpenSetCategoricalAttribute): - per_col_deltas[col] = thresholding_delta / num_open_set + per_col_deltas[col] = delta / num_open_set else: per_col_deltas[col] = 0.0 return per_col_deltas @@ -453,27 +450,20 @@ def configure( ) -> TabularMechanism: """Returns a calibrated mechanism configured with the given privacy budget. - Splits the budget additively, just as it does for ``budget``: + Splits the budget additively: - ``init_budget_fraction`` of ``budget`` goes to per-column initializers (split evenly, including a total-count mechanism); the remainder goes to the discrete mechanism. - - ``init_budget_fraction`` of ``delta`` is reserved for open-set partition - selection (split evenly across open-set columns); the remaining delta is - unused by pure-zCDP sub-mechanisms. - - When ``calibrate(epsilon, delta)`` is called, the base class binary search - passes the guarantee delta here. Because the thresholding delta is honestly - reported in the composite ``dp_event``, the binary search automatically - ensures the overall (epsilon, delta) guarantee is tight. + - ``delta`` is split evenly across open-set columns for partition selection + (pure-zCDP sub-mechanisms do not consume ``delta``). Args: schema: Dataset schema or mapping from column names to attribute domain. budget: The privacy budget. - delta: Overall approximate DP delta for the mechanism. A fraction - (``init_budget_fraction``) is allocated to partition selection for - open-set columns. Must be positive when open-set categorical attributes - are present. + delta: Approximate DP delta allocated to partition selection for open-set + columns (split evenly across open-set columns). Must be positive when + open-set categorical attributes are present. max_records_per_user: Assumed upper bound on the number of records a single user contributes. Values greater than 1 scale the added noise (and mechanism sensitivity) to provide user-level rather than diff --git a/dpsynth/relational/synthesizer.py b/dpsynth/relational/synthesizer.py index 02217b57..80eeb059 100644 --- a/dpsynth/relational/synthesizer.py +++ b/dpsynth/relational/synthesizer.py @@ -426,7 +426,6 @@ def _create_table_initializers( def _compute_table_col_deltas( domains: Mapping[str, domain.Schema], delta: float, - init_budget_fraction: float, ) -> dict[str, dict[str, float]]: """Splits thresholding delta additively across open-set columns in all tables. @@ -437,7 +436,6 @@ def _compute_table_col_deltas( Args: domains: Mapping from table names to per-column AttributeType schemas. delta: Total DP delta for partition selection thresholding. - init_budget_fraction: Fraction of delta allocated to column initialization. Returns: A nested mapping from table name and column name to its allocated delta. @@ -448,8 +446,7 @@ def _compute_table_col_deltas( Formal Guarantees: - Only open-set categorical attributes consume delta - Categorical and numerical attributes operate under pure zCDP (delta = 0.0) - - Sum of per-column deltas across all tables equals init_budget_fraction * - delta. + - Sum of per-column deltas across all tables equals delta. - Invariance: If no open-set columns exist, all per-column deltas are 0.0. """ num_open_set = 0 @@ -462,8 +459,7 @@ def _compute_table_col_deltas( 'delta must be positive when open-set categorical attributes are' ' present. It is used for Gaussian partition selection.' ) - thresholding_delta = init_budget_fraction * delta - per_col_delta = thresholding_delta / num_open_set if num_open_set > 0 else 0.0 + per_col_delta = delta / num_open_set if num_open_set > 0 else 0.0 return { table: { col: ( @@ -1259,11 +1255,7 @@ def configure( hierarchy, max_records_per_user=max_records_per_user ) - per_col_deltas = _compute_table_col_deltas( - domains, - delta=delta, - init_budget_fraction=self.init_budget_fraction, - ) + per_col_deltas = _compute_table_col_deltas(domains, delta=delta) inits = ( self.initializers if self.initializers is not None diff --git a/tests/data_generation_v3_test.py b/tests/data_generation_v3_test.py index 67e9f14a..6b9f2671 100644 --- a/tests/data_generation_v3_test.py +++ b/tests/data_generation_v3_test.py @@ -239,6 +239,29 @@ def test_calibrate_domain_positional_only(self): with self.assertRaises(TypeError): dpsynth.calibrate(config, domain=domains, epsilon=1.0, delta=1e-5) # pyrefly: ignore[unexpected-keyword] + def test_calibrate_delta_split(self): + domains = { + 'A': domain.OpenSetCategoricalAttribute(), + 'B': domain.OpenSetCategoricalAttribute(), + 'C': domain.CategoricalAttribute(possible_values=['x', 'y']), + } + config = TabularConfig() + calibrated = dpsynth.calibrate(config, domains, epsilon=1.0, delta=1e-5) + self.assertAlmostEqual(calibrated.initializers['A'].delta, 2.5e-6) + self.assertAlmostEqual(calibrated.initializers['B'].delta, 2.5e-6) + + custom = dpsynth.calibrate( + config, domains, epsilon=1.0, delta=1e-5, delta_split=0.2 + ) + self.assertAlmostEqual(custom.initializers['A'].delta, 1e-6) + self.assertAlmostEqual(custom.initializers['B'].delta, 1e-6) + + for bad_split in (0.0, 1.0, -0.1, 1.5): + with self.assertRaisesRegex(ValueError, 'delta_split must be in'): + dpsynth.calibrate( + config, domains, epsilon=1.0, delta=1e-5, delta_split=bad_split + ) + @parameterized.product( sentinel=[np.nan, None], clip_to_range=[True, False], diff --git a/tests/local_mode/initialization_test.py b/tests/local_mode/initialization_test.py index f621bd7f..3bb5c48d 100644 --- a/tests/local_mode/initialization_test.py +++ b/tests/local_mode/initialization_test.py @@ -288,6 +288,13 @@ def test_dp_event(self): ) self.assertEqual(event.events[1].delta, 1e-5) + def test_calibrate(self): + attr = domain.OpenSetCategoricalAttribute(default_value='') + config = initialization.OpenSetInitializerConfig() + calibrated = config.calibrate(attr, epsilon=1.0, delta=1e-5) + self.assertAlmostEqual(calibrated.delta, 5e-6) + self.assertGreater(calibrated.sigma, 0.0) + def test_call_noiseless(self): attr = domain.OpenSetCategoricalAttribute(default_value='') rng = np.random.default_rng(42) diff --git a/tests/relational/synthesizer_test.py b/tests/relational/synthesizer_test.py index a77b2fde..f0d9ff18 100644 --- a/tests/relational/synthesizer_test.py +++ b/tests/relational/synthesizer_test.py @@ -97,14 +97,11 @@ def test_compute_table_col_deltas_with_open_set(self): 'gender': domain.CategoricalAttribute(possible_values=['M', 'F']), }, } - deltas = synthesizer._compute_table_col_deltas( - domains, delta=1e-4, init_budget_fraction=0.2 - ) - # Total thresholding delta = 0.2 * 1e-4 = 2e-5. - # 2 open-set columns -> each gets 1e-5. - self.assertAlmostEqual(deltas['Household']['tags'], 1e-5) + deltas = synthesizer._compute_table_col_deltas(domains, delta=1e-4) + # 2 open-set columns -> each gets 5e-5. + self.assertAlmostEqual(deltas['Household']['tags'], 5e-5) self.assertEqual(deltas['Household']['income'], 0.0) - self.assertAlmostEqual(deltas['Person']['hobbies'], 1e-5) + self.assertAlmostEqual(deltas['Person']['hobbies'], 5e-5) self.assertEqual(deltas['Person']['gender'], 0.0) def test_compute_table_col_deltas_no_open_set(self): @@ -114,9 +111,7 @@ def test_compute_table_col_deltas_no_open_set(self): 'region': domain.CategoricalAttribute(possible_values=['U', 'R']), }, } - deltas = synthesizer._compute_table_col_deltas( - domains, delta=0.0, init_budget_fraction=0.1 - ) + deltas = synthesizer._compute_table_col_deltas(domains, delta=0.0) self.assertEqual(deltas['Household']['income'], 0.0) self.assertEqual(deltas['Household']['region'], 0.0) @@ -127,9 +122,7 @@ def test_compute_table_col_deltas_missing_delta_raises(self): }, } with self.assertRaisesRegex(ValueError, 'delta must be positive'): - synthesizer._compute_table_col_deltas( - domains, delta=0.0, init_budget_fraction=0.1 - ) + synthesizer._compute_table_col_deltas(domains, delta=0.0) def test_dp_event_composition(self): # Setup calibrated initializers for 2 tables.