diff --git a/.gitignore b/.gitignore index 6ef09d7..9b31389 100644 --- a/.gitignore +++ b/.gitignore @@ -2,3 +2,4 @@ venv/ **/venv/ *.parquet *.zip +*.csv diff --git a/README.md b/README.md index 5b5f144..3c203c2 100644 --- a/README.md +++ b/README.md @@ -1 +1,3 @@ # Budget Allocation in Differential Privacy + +O experimento está em experiment.py diff --git a/__pycache__/data.cpython-314.pyc b/__pycache__/data.cpython-314.pyc new file mode 100644 index 0000000..31decf1 Binary files /dev/null and b/__pycache__/data.cpython-314.pyc differ diff --git a/__pycache__/dp.cpython-314.pyc b/__pycache__/dp.cpython-314.pyc new file mode 100644 index 0000000..113fd1b Binary files /dev/null and b/__pycache__/dp.cpython-314.pyc differ diff --git a/__pycache__/experiment.cpython-314.pyc b/__pycache__/experiment.cpython-314.pyc new file mode 100644 index 0000000..de9566c Binary files /dev/null and b/__pycache__/experiment.cpython-314.pyc differ diff --git a/__pycache__/t_tests.cpython-314.pyc b/__pycache__/t_tests.cpython-314.pyc index 9b75777..c453cb5 100644 Binary files a/__pycache__/t_tests.cpython-314.pyc and b/__pycache__/t_tests.cpython-314.pyc differ diff --git a/bud.py b/bud.py new file mode 100644 index 0000000..cbb59a7 --- /dev/null +++ b/bud.py @@ -0,0 +1,3 @@ + + +def get_Bud(): diff --git a/data.py b/data.py index 1b601d4..20daa39 100644 --- a/data.py +++ b/data.py @@ -51,10 +51,10 @@ def get_regions(df: pd.DataFrame, C: float): region_df[carteira_assinada] == 1, renda_norm ].std(ddof=0), "sens_count": 1.0, - "sens_formal_mean": C / (n_formal - 1), - "sens_formal_std": C / np.sqrt(n_formal - 1), - "sens_informal_mean": C / (n_informal - 1), - "sens_informal_std": C / np.sqrt(n_informal - 1), + "sens_formal_mean": 1 / (n_formal - 1), + "sens_formal_std": 1 / np.sqrt(n_formal - 1), + "sens_informal_mean": 1 / (n_informal - 1), + "sens_informal_std": 1 / np.sqrt(n_informal - 1), } return regions diff --git a/experiment.py b/experiment.py new file mode 100644 index 0000000..1d31953 --- /dev/null +++ b/experiment.py @@ -0,0 +1,152 @@ +import sys +from data import get_regions, clip_and_normalize +import pandas as pd +from t_tests import t +from dp import epsilon_dp + +FILE = "./data/pnad_trimestral_trimestre_012026.parquet" +C = 46_366 # teto constitucional (salário de ministros da suprema corte) + + +def generate_sequences(length=12, total=12, granularity=0.5): + units = int(total / granularity) + min_value = 1 # 0.5 na escala de granularidade + + def generate(position, remaining, sequence): + if position == length - 1: + if remaining >= min_value: + yield tuple(sequence + [remaining]) + return + + # Precisamos deixar pelo menos 1 para cada posição restante + max_value = remaining - (length - position - 1) * min_value + + for value in range(min_value, max_value + 1): + yield from generate(position + 1, remaining - value, sequence + [value]) + + for sequence in generate(0, units, []): + yield tuple(value * granularity for value in sequence) + + +# comparação a ser feita: sudeste informal vs nordeste informal +def get_informal_sta(regions): + return ( + regions["Sudeste"]["informal_count"], + regions["Sudeste"]["informal_mean"], + regions["Sudeste"]["informal_std"], + regions["Nordeste"]["informal_count"], + regions["Nordeste"]["informal_mean"], + regions["Nordeste"]["informal_std"], + ) + + +def get_informal_sen(regions): + return ( + regions["Sudeste"]["sens_count"], + regions["Sudeste"]["sens_informal_mean"], + regions["Sudeste"]["sens_informal_std"], + regions["Nordeste"]["sens_count"], + regions["Nordeste"]["sens_informal_mean"], + regions["Nordeste"]["sens_informal_std"], + ) + + +# comparação a ser feita: sudeste formal vs nordeste formal +def get_formal_sta(regions): + return ( + regions["Sudeste"]["formal_count"], + regions["Sudeste"]["formal_mean"], + regions["Sudeste"]["formal_std"], + regions["Nordeste"]["formal_count"], + regions["Nordeste"]["formal_mean"], + regions["Nordeste"]["formal_std"], + ) + + +def get_formal_sen(regions): + return ( + regions["Sudeste"]["sens_count"], + regions["Sudeste"]["sens_formal_mean"], + regions["Sudeste"]["sens_formal_std"], + regions["Nordeste"]["sens_count"], + regions["Nordeste"]["sens_formal_mean"], + regions["Nordeste"]["sens_formal_std"], + ) + + +def us(bud_sequence, sen_sequence): + result = 0 + for i in range(12): + result += sen_sequence[i] / bud_sequence[i] + return result + + +def ue(sta_true, bud, sen): + result = [0.0, 0.0] + for i in range(1000): + sta_noisy = [] + for i in range(12): + sta_noisy.append(epsilon_dp(sta_true[i], bud[i], sen[i])) + + t_true_informal = t(*sta_true[0:6]) + t_true_formal = t(*sta_true[6:12]) + t_noisy_informal = t(*sta_noisy[0:6]) + t_noisy_formal = t(*sta_noisy[6:12]) + result[0] += abs(t_true_informal - t_noisy_informal) + result[1] += abs(t_true_formal - t_noisy_formal) + + result[0] = result[0] / 1000 + result[1] = result[1] / 1000 + + return result + + +df = pd.read_parquet(FILE) +df = clip_and_normalize(df, C) +regions = get_regions(df, C) + +sequences = [] +if sys.argv[1] == "real": + sequences = generate_sequences() +elif sys.argv[1] == "teste": + sequences = [ + [1 for _ in range(12)], + [0.5, 0.5, 0.5, 0.5, 0.5, 0.5, 1.5, 1.5, 1.5, 1.5, 1.5, 1.5], + [2 for _ in range(12)], + ] + +informal_sta = get_informal_sta(regions) +formal_sta = get_formal_sta(regions) +informal_sen = get_informal_sen(regions) +formal_sen = get_formal_sen(regions) + +print(informal_sen) +print(formal_sen) + +results = [] +best_metric = sys.float_info.max + +for bud in sequences: + metric = 0 + sta = informal_sta + formal_sta + sen = informal_sen + formal_sen + us_result = us(bud, sen) + ue_result = ue(sta, bud, sen) + metric += us_result + metric += ue_result[0] + ue_result[1] + metric = metric / 14.0 + print(metric) + results.append([*bud, metric]) + if metric < best_metric: + best_metric = metric + +print(f"melhor metrica: {best_metric}") + +# gera um dataframe do pandas com cada valor de bud de cada sequência, seguido do resultado da metrica para aquela sequência +cols = [f"bud_{i}" for i in range(1, 13)] +cols.append("metric") + +result_df = pd.DataFrame(results, columns=cols) +result_df.to_csv("result.csv") + +print("Dados salvos em result.csv")