diff --git a/.gitignore b/.gitignore index 9b31389..4252190 100644 --- a/.gitignore +++ b/.gitignore @@ -3,3 +3,4 @@ venv/ *.parquet *.zip *.csv +__pycache__/ diff --git a/README.md b/README.md index 3c203c2..6b13c33 100644 --- a/README.md +++ b/README.md @@ -1,3 +1,44 @@ -# Budget Allocation in Differential Privacy +# Experiment: PREDICTIVE METRIC FOR OPTIMAL BUDGET ALLOCATION IN DIFFERENTIAL PRIVACY -O experimento está em experiment.py +Based on the [original experiment](https://github.com/conseg/TheImpactofDifferentialPrivacyondatautilityinfundamentalmathematicaloperations), but using a real life dataset. + +Key difference: sensitivity calculation to satisfy Differential Privacy. + +## The Dataset + +PNAD Contínua (IBGE - Brasil) - 2026 (primeiro trimestre) + +## The Experiment + +Objective: evaluate the metric on a t-test equation to compare Northeast and Southeast income (informal vs formal workers) + +Steps: + +1. Clip and normalize dataset (C = 46_366, the constitutional salary cap, which is set at the salary of Supreme Court justices). Must be a value taken from the real world, not from the dataset. +2. Calculate necessary statistics and their sensitivities +3. For each possible budget allocation with total epsilon=12, granularity=0.5, epsilon > 0 (1,352,078 possibilities): evaluate the metric for the statistics and the budget allocation sequence +4. Save all possible scores and budget allocation sequences to csv for further analysis, and print the best metric score. + +## Estimated Experiment Execution Time + +To reproduce (in a Linux shell): + +1. `time python ./experiment.py real` +2. Send an interrupt (Ctrl+C) when N exceeds 500. +3. To find the number of hours, based on the last N and the elapsed time (t): `((1352078*t)/n)/3600` + +Laptop (Intel Core i5 1135G7 - 8GB RAM) +PC (AMD Ryzen 7 8700G - 32GB RAM) + +Laptop: 514 iterations in 12.13s +PC: 493 iterations in 8.11s + +Estimated total time on the laptop: 8h51m +Estimated total time on the PC: 6h10m + +## Tasks + +1. Provide a way to run the experiment in parts, to allow parallel execution on multiple CPUs. +2. Run the experiment to evaluate the metric. +3. Combine the resulting CSVs into one. +4. Analyze the csv to find notable metric scores for specific sequences. diff --git a/__pycache__/data.cpython-314.pyc b/__pycache__/data.cpython-314.pyc deleted file mode 100644 index 31decf1..0000000 Binary files a/__pycache__/data.cpython-314.pyc and /dev/null differ diff --git a/__pycache__/dp.cpython-314.pyc b/__pycache__/dp.cpython-314.pyc deleted file mode 100644 index 113fd1b..0000000 Binary files a/__pycache__/dp.cpython-314.pyc and /dev/null differ diff --git a/__pycache__/experiment.cpython-314.pyc b/__pycache__/experiment.cpython-314.pyc deleted file mode 100644 index de9566c..0000000 Binary files a/__pycache__/experiment.cpython-314.pyc and /dev/null differ diff --git a/__pycache__/get_statistics.cpython-314.pyc b/__pycache__/get_statistics.cpython-314.pyc deleted file mode 100644 index 47cc162..0000000 Binary files a/__pycache__/get_statistics.cpython-314.pyc and /dev/null differ diff --git a/__pycache__/t_tests.cpython-314.pyc b/__pycache__/t_tests.cpython-314.pyc deleted file mode 100644 index c453cb5..0000000 Binary files a/__pycache__/t_tests.cpython-314.pyc and /dev/null differ diff --git a/experiment.py b/experiment.py index 1d31953..cf5615a 100644 --- a/experiment.py +++ b/experiment.py @@ -126,7 +126,9 @@ print(formal_sen) results = [] best_metric = sys.float_info.max +n = 0 for bud in sequences: + n += 1 metric = 0 sta = informal_sta + formal_sta sen = informal_sen + formal_sen @@ -135,7 +137,7 @@ for bud in sequences: metric += us_result metric += ue_result[0] + ue_result[1] metric = metric / 14.0 - print(metric) + print(f"n={n}; metric={metric}") results.append([*bud, metric]) if metric < best_metric: best_metric = metric