|
| 1 | +"""SEFR: a single-pass, linear-time classifier. |
| 2 | +
|
| 3 | +Reference: |
| 4 | + Keshavarz, Saniee Abadeh, Rawassizadeh (2020). |
| 5 | + "SEFR: A Fast Linear-Time Classifier for Ultra-Low Power Devices". |
| 6 | + https://arxiv.org/abs/2006.04620 |
| 7 | +
|
| 8 | +SEFR derives one weight per feature plus a bias from class-conditional feature |
| 9 | +means, so fitting is a single O(n_samples * n_features) pass with no iterative |
| 10 | +optimization. The trained model is ``n_features + 1`` floats. That makes it the |
| 11 | +cheapest learner in the FLAML portfolio and the only one that reliably returns a |
| 12 | +model within a very small ``time_budget`` on large data. |
| 13 | +
|
| 14 | +It is implemented here in NumPy rather than pulled in as a dependency: the whole |
| 15 | +algorithm is a handful of array operations, and FLAML's only required dependency |
| 16 | +is NumPy. |
| 17 | +""" |
| 18 | + |
| 19 | +from __future__ import annotations |
| 20 | + |
| 21 | +import numpy as np |
| 22 | + |
| 23 | +try: |
| 24 | + from scipy.sparse import issparse |
| 25 | +except ImportError: |
| 26 | + |
| 27 | + def issparse(X): |
| 28 | + return False |
| 29 | + |
| 30 | + |
| 31 | +try: |
| 32 | + from sklearn.base import BaseEstimator as SKLearnBaseEstimator |
| 33 | + from sklearn.base import ClassifierMixin |
| 34 | + from sklearn.ensemble import AdaBoostClassifier |
| 35 | + from sklearn.linear_model import LogisticRegression |
| 36 | +except ImportError as e: |
| 37 | + print(f"scikit-learn is required for SEFREstimator. Please install it; error: {e}") |
| 38 | + |
| 39 | +from flaml import tune |
| 40 | +from flaml.automl.model import SKLearnEstimator |
| 41 | +from flaml.automl.task import Task |
| 42 | + |
| 43 | +_EPS = 1e-7 |
| 44 | + |
| 45 | + |
| 46 | +def _column_weighted_mean(X, weights, total): |
| 47 | + """Weighted column means of ``X``; works for dense arrays and sparse matrices.""" |
| 48 | + return np.asarray(X.T.dot(weights)).ravel() / total |
| 49 | + |
| 50 | + |
| 51 | +def _to_dense_row(values): |
| 52 | + """Collapse the result of a column-wise reduction to a 1-D ndarray.""" |
| 53 | + if issparse(values): |
| 54 | + values = values.toarray() |
| 55 | + return np.asarray(values).ravel() |
| 56 | + |
| 57 | + |
| 58 | +class SEFRClassifier(ClassifierMixin, SKLearnBaseEstimator): |
| 59 | + """Scalable, Efficient and Fast classifieR (SEFR). |
| 60 | +
|
| 61 | + Binary targets are fit with the closed form of the paper (Eqs. 3-9); |
| 62 | + multiclass targets use the one-vs-rest scheme of Sec. 3.4. |
| 63 | +
|
| 64 | + Parameters |
| 65 | + ---------- |
| 66 | + scaling : {"minmax", "maxabs", "none"}, default="minmax" |
| 67 | + SEFR's weight formula assumes non-negative features, and FLAML does not |
| 68 | + scale features anywhere in its pipeline, so scaling belongs to the |
| 69 | + estimator. Sparse input falls back to "maxabs", which preserves sparsity. |
| 70 | + class_weight : {"none", "balanced"}, default="none" |
| 71 | + "balanced" reweights samples inversely to class frequency before the |
| 72 | + means of Eq. 3-4 are taken. |
| 73 | + threshold : {"sefr", "balanced"}, default="sefr" |
| 74 | + "sefr" is the class-count-weighted score average of Eq. 9. "balanced" |
| 75 | + is the unweighted midpoint of the two class score means. |
| 76 | + threshold_shift : float, default=0.0 |
| 77 | + Shifts the bias by this many standard deviations of the training scores. |
| 78 | + Only affects `predict`, not the ranking produced by `decision_function`. |
| 79 | + calibration : {"platt", "sigmoid"}, default="platt" |
| 80 | + SEFR produces margins, not probabilities. "sigmoid" squashes the margin |
| 81 | + by its training standard deviation; "platt" fits a one-dimensional |
| 82 | + logistic regression on the training margins. Both are monotone, so the |
| 83 | + choice does not affect ROC AUC, but it matters a great deal for |
| 84 | + log_loss, which is FLAML's default multiclass metric. |
| 85 | + eps : float, default=1e-7 |
| 86 | + Stabilizer for the denominator of Eq. 5. |
| 87 | + """ |
| 88 | + |
| 89 | + def __init__( |
| 90 | + self, |
| 91 | + scaling="minmax", |
| 92 | + class_weight="none", |
| 93 | + threshold="sefr", |
| 94 | + threshold_shift=0.0, |
| 95 | + calibration="platt", |
| 96 | + eps=_EPS, |
| 97 | + ): |
| 98 | + self.scaling = scaling |
| 99 | + self.class_weight = class_weight |
| 100 | + self.threshold = threshold |
| 101 | + self.threshold_shift = threshold_shift |
| 102 | + self.calibration = calibration |
| 103 | + self.eps = eps |
| 104 | + |
| 105 | + def _fit_scaler(self, X): |
| 106 | + scaling = self.scaling |
| 107 | + if issparse(X) and scaling == "minmax": |
| 108 | + # subtracting a per-column minimum would densify the matrix |
| 109 | + scaling = "maxabs" |
| 110 | + if scaling == "minmax": |
| 111 | + self.offset_ = _to_dense_row(X.min(axis=0)) |
| 112 | + spread = _to_dense_row(X.max(axis=0)) - self.offset_ |
| 113 | + elif scaling == "maxabs": |
| 114 | + self.offset_ = None |
| 115 | + spread = _to_dense_row(abs(X).max(axis=0)) |
| 116 | + else: |
| 117 | + self.offset_ = None |
| 118 | + spread = None |
| 119 | + if spread is not None: |
| 120 | + spread[spread == 0] = 1.0 |
| 121 | + self.spread_ = spread |
| 122 | + |
| 123 | + def _scale(self, X): |
| 124 | + if self.spread_ is None: |
| 125 | + return X |
| 126 | + if issparse(X): |
| 127 | + # multiply() keeps the matrix sparse; dividing by a dense row would not |
| 128 | + return X.multiply(1.0 / self.spread_).tocsr() |
| 129 | + if self.offset_ is None: |
| 130 | + return X / self.spread_ |
| 131 | + return (X - self.offset_) / self.spread_ |
| 132 | + |
| 133 | + def _fit_head(self, X, is_positive, sample_weight): |
| 134 | + """Return (coef, bias, margins) for one binary problem.""" |
| 135 | + negative = ~is_positive |
| 136 | + w_pos, w_neg = sample_weight[is_positive], sample_weight[negative] |
| 137 | + sum_pos, sum_neg = w_pos.sum(), w_neg.sum() |
| 138 | + |
| 139 | + avg_pos = _column_weighted_mean(X[is_positive], w_pos, sum_pos) |
| 140 | + avg_neg = _column_weighted_mean(X[negative], w_neg, sum_neg) |
| 141 | + coef = (avg_pos - avg_neg) / (avg_pos + avg_neg + self.eps) |
| 142 | + |
| 143 | + scores = np.asarray(X @ coef).ravel() |
| 144 | + score_pos = np.average(scores[is_positive], weights=w_pos) |
| 145 | + score_neg = np.average(scores[negative], weights=w_neg) |
| 146 | + if self.threshold == "balanced": |
| 147 | + bias = 0.5 * (score_pos + score_neg) |
| 148 | + else: |
| 149 | + bias = (sum_neg * score_pos + sum_pos * score_neg) / (sum_neg + sum_pos) |
| 150 | + if self.threshold_shift: |
| 151 | + bias += self.threshold_shift * (scores.std() or 1.0) |
| 152 | + return coef, bias, scores - bias |
| 153 | + |
| 154 | + def _fit_calibrator(self, margins, target): |
| 155 | + if self.calibration == "sigmoid": |
| 156 | + return float(margins.std() or 1.0) |
| 157 | + calibrator = LogisticRegression(solver="lbfgs") |
| 158 | + calibrator.fit(margins.reshape(-1, 1), target) |
| 159 | + return calibrator |
| 160 | + |
| 161 | + @staticmethod |
| 162 | + def _apply_calibrator(calibrator, margins): |
| 163 | + if isinstance(calibrator, float): |
| 164 | + return 1.0 / (1.0 + np.exp(-np.clip(margins / calibrator, -30, 30))) |
| 165 | + return calibrator.predict_proba(margins.reshape(-1, 1))[:, 1] |
| 166 | + |
| 167 | + def fit(self, X, y, sample_weight=None): |
| 168 | + if not issparse(X): |
| 169 | + X = np.asarray(X, dtype=np.float64) |
| 170 | + y = np.asarray(y).ravel() |
| 171 | + if X.ndim != 2: |
| 172 | + raise ValueError(f"X must be 2-D, got shape {X.shape}") |
| 173 | + |
| 174 | + self.classes_ = np.unique(y) |
| 175 | + self.n_features_in_ = X.shape[1] |
| 176 | + if self.classes_.size < 2: |
| 177 | + raise ValueError("SEFR needs at least two classes in the training data") |
| 178 | + |
| 179 | + if sample_weight is None: |
| 180 | + sample_weight = np.ones(y.shape[0], dtype=np.float64) |
| 181 | + else: |
| 182 | + sample_weight = np.asarray(sample_weight, dtype=np.float64).ravel() |
| 183 | + if self.class_weight == "balanced": |
| 184 | + counts = {c: sample_weight[y == c].sum() for c in self.classes_} |
| 185 | + scale = y.shape[0] / (self.classes_.size * np.array([counts[c] for c in self.classes_])) |
| 186 | + lookup = dict(zip(self.classes_, scale)) |
| 187 | + sample_weight = sample_weight * np.array([lookup[label] for label in y]) |
| 188 | + |
| 189 | + self._fit_scaler(X) |
| 190 | + X = self._scale(X) |
| 191 | + |
| 192 | + heads = [self.classes_[1]] if self.classes_.size == 2 else list(self.classes_) |
| 193 | + coefs, biases, self.calibrators_ = [], [], [] |
| 194 | + for label in heads: |
| 195 | + is_positive = y == label |
| 196 | + coef, bias, margins = self._fit_head(X, is_positive, sample_weight) |
| 197 | + coefs.append(coef) |
| 198 | + biases.append(bias) |
| 199 | + self.calibrators_.append(self._fit_calibrator(margins, is_positive.astype(int))) |
| 200 | + self.coef_ = np.vstack(coefs) |
| 201 | + self.intercept_ = np.array(biases) |
| 202 | + return self |
| 203 | + |
| 204 | + def decision_function(self, X): |
| 205 | + if not issparse(X): |
| 206 | + X = np.asarray(X, dtype=np.float64) |
| 207 | + scores = np.asarray(self._scale(X) @ self.coef_.T) - self.intercept_ |
| 208 | + return scores.ravel() if self.coef_.shape[0] == 1 else scores |
| 209 | + |
| 210 | + def predict(self, X): |
| 211 | + scores = self.decision_function(X) |
| 212 | + if self.coef_.shape[0] == 1: |
| 213 | + return self.classes_[(scores > 0).astype(int)] |
| 214 | + return self.classes_[np.argmax(scores, axis=1)] |
| 215 | + |
| 216 | + def predict_proba(self, X): |
| 217 | + scores = self.decision_function(X) |
| 218 | + if self.coef_.shape[0] == 1: |
| 219 | + positive = self._apply_calibrator(self.calibrators_[0], scores) |
| 220 | + return np.column_stack([1.0 - positive, positive]) |
| 221 | + proba = np.column_stack([self._apply_calibrator(cal, scores[:, i]) for i, cal in enumerate(self.calibrators_)]) |
| 222 | + total = proba.sum(axis=1, keepdims=True) |
| 223 | + total[total == 0] = 1.0 |
| 224 | + return proba / total |
| 225 | + |
| 226 | + |
| 227 | +class SEFREstimator(SKLearnEstimator): |
| 228 | + """The class for tuning SEFR.""" |
| 229 | + |
| 230 | + @classmethod |
| 231 | + def search_space(cls, **params) -> dict: |
| 232 | + return { |
| 233 | + "scaling": { |
| 234 | + "domain": tune.choice(["minmax", "maxabs"]), |
| 235 | + "init_value": "minmax", |
| 236 | + }, |
| 237 | + "class_weight": { |
| 238 | + "domain": tune.choice(["none", "balanced"]), |
| 239 | + "init_value": "none", |
| 240 | + }, |
| 241 | + "threshold": { |
| 242 | + "domain": tune.choice(["sefr", "balanced"]), |
| 243 | + "init_value": "sefr", |
| 244 | + }, |
| 245 | + "threshold_shift": { |
| 246 | + "domain": tune.uniform(lower=-1.0, upper=1.0), |
| 247 | + "init_value": 0.0, |
| 248 | + }, |
| 249 | + "calibration": { |
| 250 | + "domain": tune.choice(["platt", "sigmoid"]), |
| 251 | + "init_value": "platt", |
| 252 | + }, |
| 253 | + } |
| 254 | + |
| 255 | + @classmethod |
| 256 | + def cost_relative2lgbm(cls) -> float: |
| 257 | + return 1.0 |
| 258 | + |
| 259 | + def config2params(self, config: dict) -> dict: |
| 260 | + params = super().config2params(config) |
| 261 | + params.pop("n_jobs", None) |
| 262 | + params.pop("random_state", None) |
| 263 | + return params |
| 264 | + |
| 265 | + def __init__(self, task: Task = "binary", **config): |
| 266 | + super().__init__(task, **config) |
| 267 | + assert self._task.is_classification(), "SEFR is for classification tasks only" |
| 268 | + self.estimator_class = SEFRClassifier |
| 269 | + |
| 270 | + |
| 271 | +class SEFRBoostEstimator(SKLearnEstimator): |
| 272 | + """The class for tuning AdaBoost over SEFR base learners. |
| 273 | +
|
| 274 | + SEFR has no hyperparameters of its own to trade accuracy against cost, so on |
| 275 | + its own it gives the search very little to do. Boosting it keeps the |
| 276 | + single-pass base learner while giving FLAML a real, cheap-to-traverse search |
| 277 | + space. |
| 278 | + """ |
| 279 | + |
| 280 | + ITER_HP = "n_estimators" |
| 281 | + DEFAULT_ITER = 50 |
| 282 | + |
| 283 | + @classmethod |
| 284 | + def search_space(cls, data_size, **params) -> dict: |
| 285 | + upper = max(5, min(1024, int(data_size[0]))) |
| 286 | + space = { |
| 287 | + "n_estimators": { |
| 288 | + "domain": tune.lograndint(lower=4, upper=upper), |
| 289 | + "init_value": 4, |
| 290 | + "low_cost_init_value": 4, |
| 291 | + }, |
| 292 | + "learning_rate": { |
| 293 | + "domain": tune.loguniform(lower=1 / 1024, upper=1.0), |
| 294 | + "init_value": 0.1, |
| 295 | + }, |
| 296 | + } |
| 297 | + space.update( |
| 298 | + { |
| 299 | + key: value |
| 300 | + for key, value in SEFREstimator.search_space(**params).items() |
| 301 | + if key in ("scaling", "class_weight") |
| 302 | + } |
| 303 | + ) |
| 304 | + return space |
| 305 | + |
| 306 | + @classmethod |
| 307 | + def cost_relative2lgbm(cls) -> float: |
| 308 | + return 5.0 |
| 309 | + |
| 310 | + def config2params(self, config: dict) -> dict: |
| 311 | + params = super().config2params(config) |
| 312 | + params.pop("n_jobs", None) |
| 313 | + base_params = {key: params.pop(key) for key in ("scaling", "class_weight") if key in params} |
| 314 | + params["estimator"] = SEFRClassifier(**base_params) |
| 315 | + if "random_state" not in params: |
| 316 | + params["random_state"] = 24092023 |
| 317 | + return params |
| 318 | + |
| 319 | + def __init__(self, task: Task = "binary", **config): |
| 320 | + super().__init__(task, **config) |
| 321 | + assert self._task.is_classification(), "SEFR is for classification tasks only" |
| 322 | + self.estimator_class = AdaBoostClassifier |
0 commit comments