-
Notifications
You must be signed in to change notification settings - Fork 36
Expand file tree
/
Copy pathstrategies.py
More file actions
217 lines (188 loc) · 11 KB
/
Copy pathstrategies.py
File metadata and controls
217 lines (188 loc) · 11 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
"""The five mitigation strategies (S0-S4) - Layer 2 of the benchmark harness.
Given a manifest's core/proxy/protected feature partition, builds the
train/test feature set (and, for S3/S4, fits the fairlearn mitigator) for
five strategies of increasing sophistication:
S0 baseline - every feature; protected attributes and
proxies included as model input. The "do
nothing" reference.
S1 unawareness - protected attribute dropped from the
model input; proxies retained. The naive
fix people assume works.
S2 unawareness_proxy_removal - protected attribute AND proxies dropped
from the model input (the classic
fair.py fix every existing audit uses).
S3 in_processing - fairlearn ExponentiatedGradient: trains
an ensemble of base estimators under a
fairness constraint (see
FAIRNESS_CONSTRAINT).
S4 post_processing - fairlearn ThresholdOptimizer: fits one
base estimator, then finds a per-group
decision threshold over its predicted
probabilities that satisfies the same
constraint.
S3 and S4 use the SAME reduced feature set as S2 (core features only) - so a
residual gap there isn't explained by "the model could still see the
protected attribute or a proxy". The protected attribute is never deleted
from the working dataset (faircode.benchmark keeps it in `protected_masks`
throughout); for S3/S4 it is passed to fairlearn as `sensitive_features`
instead of as a column of X. That is the whole point of comparing S1/S2
against S3/S4: S1/S2 remove the protected attribute's influence by deleting
columns from the model input; S3/S4 remove its influence a different way,
via a fairness constraint, while seeing the identical feature set S2 sees.
Showing that S3/S4 hit roughly the same floor S2 does is what turns "here is
our fix" into "here is the residual floor even stronger tools can't clear".
"""
from __future__ import annotations
import numpy as np
import pandas as pd
import pandas.api.types as pdt
from fairlearn.postprocessing import ThresholdOptimizer
from fairlearn.reductions import DemographicParity, EqualizedOdds, ExponentiatedGradient
from sklearn.model_selection import train_test_split
STRATEGIES = ("baseline", "unawareness", "unawareness_proxy_removal", "in_processing", "post_processing")
# Every audit is run under the SAME fairness constraint, so an in-processing/
# post-processing comparison across domains stays apples-to-apples. Flip this
# constant (not a per-manifest setting) to re-run the whole benchmark under
# the other constraint.
FAIRNESS_CONSTRAINT = "demographic_parity" # or "equalized_odds"
_REDUCTIONS_CONSTRAINTS = {
"demographic_parity": DemographicParity,
"equalized_odds": EqualizedOdds,
}
# ExponentiatedGradient deep-copies and refits the base estimator once per
# iteration - this is fairlearn's own default max_iter, used as-is rather
# than a reduced value traded off against wall-clock time (an earlier
# revision ran a reduced iteration count for faster day-to-day benchmark
# runs; bumped to the full default for the paper-quality run in 04c90e4 and
# left there since, so every reported number reflects the same convergence
# fairlearn's default is meant to reach).
EXPONENTIATED_GRADIENT_MAX_ITER = 50
def _ordinal_encode(values: "pd.Series") -> "pd.Series":
"""Integer-code a categorical column, but order the codes so any
numeric-looking category sorts by its numeric value (the rest sort
alphabetically after them).
Plain LabelEncoder sorts every category as a string, so one non-numeric
sentinel ("Unknown", "N/A") in an otherwise-numeric column - which turns
the whole column object-dtyped - silently scrambles the numeric order
("10" and "20" encode below "3"). A genuinely categorical column with no
numeric-looking values encodes identically to LabelEncoder. See #514.
"""
cats = list(pd.unique(values))
as_num = pd.to_numeric(pd.Series(cats), errors="coerce")
order = sorted(
range(len(cats)),
key=lambda i: (0, float(as_num.iloc[i]), "")
if pd.notna(as_num.iloc[i]) else (1, 0.0, str(cats[i])),
)
mapping = {cats[i]: rank for rank, i in enumerate(order)}
return values.map(mapping)
def encode_features(df: pd.DataFrame, columns: list) -> pd.DataFrame:
"""Label-encode categorical columns, pass numeric columns through.
Uniform ordinal encoding (rather than one-hot) keeps the harness code
path identical across audits regardless of a categorical column's
cardinality - several audits (e.g. Healthcare Readmission's ICD
diagnosis codes) have hundreds of categories, where one-hot encoding
would blow up the feature matrix differently per audit. Missing values
are filled (median for numeric, a sentinel category for categorical) so
every model family gets a fully-populated matrix.
A categorical column is ordinal-encoded with numeric-looking categories
ordered by value, so a stray non-numeric sentinel in an otherwise
numeric column does not scramble its ordering (#514).
"""
out = pd.DataFrame(index=df.index)
for col in columns:
series = df[col]
if pdt.is_numeric_dtype(series):
numeric = pd.to_numeric(series, errors="coerce")
median = numeric.median()
out[col] = numeric.fillna(0.0 if pd.isna(median) else median)
else:
filled = series.astype(str).fillna("__missing__")
out[col] = _ordinal_encode(filled).to_numpy()
return out
def strategy_features(strategy: str, core: list, proxies: list, protected: list) -> list:
if strategy == "baseline":
return list(dict.fromkeys(core + proxies + protected))
if strategy == "unawareness":
return list(dict.fromkeys(core + proxies))
if strategy in ("unawareness_proxy_removal", "in_processing", "post_processing"):
return list(dict.fromkeys(core))
raise ValueError(f"unknown strategy: {strategy!r}")
def fit_in_processing(base_model, X_train, y_train, sensitive_train):
"""S3 - fairlearn ExponentiatedGradient (Agarwal et al. 2018).
base_model is a fresh, unfit estimator; ExponentiatedGradient deep-copies
it internally for each iteration and returns a randomized mixture over
the resulting ensemble. sensitive_train is passed as `sensitive_features`
- never as a column of X_train - so the constraint sees group membership
without the base estimator ever training on it directly.
"""
constraint = _REDUCTIONS_CONSTRAINTS[FAIRNESS_CONSTRAINT]()
mitigator = ExponentiatedGradient(
base_model, constraints=constraint, max_iter=EXPONENTIATED_GRADIENT_MAX_ITER)
mitigator.fit(X_train, y_train, sensitive_features=sensitive_train)
return mitigator
def predict_in_processing(mitigator, X_test, random_state):
"""Predict with an S3 mitigator. Returns (y_pred, y_proba); y_proba comes
from fairlearn's internal (undocumented-but-stable) _pmf_predict - the
probability-weighted mixture over the ensemble - and falls back to None
(no AUC for this run) if a future fairlearn version removes it."""
y_pred = np.asarray(mitigator.predict(X_test, random_state=random_state)).astype(int)
try:
y_proba = np.asarray(mitigator._pmf_predict(X_test))[:, 1]
except (AttributeError, IndexError):
y_proba = None
return y_pred, y_proba
def fit_post_processing(base_model, X_train, y_train, sensitive_train,
random_state=42, calibration_size=0.3):
"""S4 - fairlearn ThresholdOptimizer (Hardt, Price & Srebro 2016 style).
base_model is fit on a FIT split of the training data; per-group
thresholds are then calibrated on a separate, held-out CALIBRATION split
of that same training data (prefit=True). Calibrating thresholds against
the same rows the base estimator was fit on - fairlearn's default
prefit=False behaviour - lets an overfit classifier's in-sample
confidence produce thresholds that don't generalize: on German Credit
Lending this produced a near-zero demographic parity gap on the
calibration data (as designed) but a gap of +0.22 on the held-out test
set, because the base RandomForest's in-sample predict_proba is far more
confident than its out-of-sample behaviour. This is the same failure
mode faircode.benchmark's predecessor threshold-search had to work
around with cross_val_predict; here it's solved by simply not
calibrating against the rows the model memorized.
"""
idx = np.arange(len(y_train))
# Stratifying on y alone isn't enough: ThresholdOptimizer.fit() requires
# every sensitive-feature group present in the CALIBRATION split to have
# both label classes represented ("degenerate labels" otherwise), and the
# base model needs both classes in its own FIT split for predict_proba to
# return two columns. Stratifying on the (label, sensitive-group) pair
# instead guarantees both splits keep at least one row of every such
# combination - confirmed empirically stable across 200+ random_state
# values - as long as every combination present has at least 2 rows.
combined = np.array([f"{y}\0{s}" for y, s in zip(y_train, sensitive_train)])
_, combo_counts = np.unique(combined, return_counts=True)
if combo_counts.min() < 2:
raise ValueError(
"fit_post_processing: not enough data to calibrate reliably - "
f"the rarest (target, sensitive-attribute) combination has only "
f"{combo_counts.min()} row(s); need at least 2 so both the fit "
"and calibration splits can see it. This audit/subgroup is too "
"sparse for S4 (post_processing) to run safely."
)
fit_idx, calib_idx = train_test_split(
idx, test_size=calibration_size, random_state=random_state, stratify=combined)
base_model.fit(X_train.iloc[fit_idx], y_train[fit_idx])
optimizer = ThresholdOptimizer(
estimator=base_model, constraints=FAIRNESS_CONSTRAINT,
predict_method="predict_proba", prefit=True)
optimizer.fit(X_train.iloc[calib_idx], y_train[calib_idx],
sensitive_features=sensitive_train[calib_idx])
return optimizer
def predict_post_processing(optimizer, X_test, sensitive_test, random_state):
"""Predict with an S4 optimizer. Needs sensitive_features at predict
time too - it must know which group's threshold to apply per row.
ThresholdOptimizer has no probability output, so y_proba is always None
(no AUC for this strategy - a hard-threshold post-processor has nothing
left to rank)."""
y_pred = np.asarray(optimizer.predict(
X_test, sensitive_features=sensitive_test, random_state=random_state)).astype(int)
return y_pred, None