diff --git a/flaml/automl/contrib/__init__.py b/flaml/automl/contrib/__init__.py index d0a356fff4..8fd2e9bed6 100644 --- a/flaml/automl/contrib/__init__.py +++ b/flaml/automl/contrib/__init__.py @@ -1 +1,2 @@ from .histgb import HistGradientBoostingEstimator +from .sefr import SEFRBoostEstimator, SEFREstimator diff --git a/flaml/automl/contrib/sefr.py b/flaml/automl/contrib/sefr.py new file mode 100644 index 0000000000..5246de42b6 --- /dev/null +++ b/flaml/automl/contrib/sefr.py @@ -0,0 +1,545 @@ +"""SEFR: a linear-time, closed-form classifier. + +Reference: + Keshavarz, Saniee Abadeh, Rawassizadeh (2020). + "SEFR: A Fast Linear-Time Classifier for Ultra-Low Power Devices". + https://arxiv.org/abs/2006.04620 + +SEFR derives one weight per feature plus a bias from class-conditional feature +means. Fitting is a closed form with no iterative optimization, and takes three +passes over the data whatever the number of classes: the feature range, the +per-class feature sums (one sparse one-hot matrix product), and the spread of +the training scores (in row chunks). Every one-vs-rest head and its Eq. 9 bias +follow from the per-class sums, so beyond the scaled copy of the input the fit +holds O(n_classes * n_features) memory, not O(n_samples * n_classes). +The optional "platt" calibration adds an iterative fit of n_classes + 1 +parameters on at most ``_CALIBRATION_MAX_ENTRIES`` training margins. + +A binary SEFR head is ``n_features + 1`` floats; the fitted classifier also +stores the per-feature scaling parameters, one head per class for multiclass +targets, and the calibration (one scale, plus one bias per class). + +It is implemented here in NumPy rather than pulled in as a dependency: the whole +algorithm is a handful of array operations, and FLAML's only required dependency +is NumPy. +""" + +from __future__ import annotations + +import inspect +import numbers + +import numpy as np + +try: + from scipy.sparse import issparse +except ImportError: + + def issparse(X): + return False + + +try: + from sklearn.base import BaseEstimator as SKLearnBaseEstimator + from sklearn.base import ClassifierMixin + from sklearn.ensemble import AdaBoostClassifier + from sklearn.utils.multiclass import check_classification_targets + from sklearn.utils.validation import check_is_fitted + + try: + from sklearn.utils.validation import validate_data as _validate_data # scikit-learn >= 1.6 + except ImportError: + + def _validate_data(estimator, **kwargs): + return estimator._validate_data(**kwargs) + +except ImportError as e: + print(f"scikit-learn is required for SEFREstimator. Please install it; error: {e}") + +from flaml import tune +from flaml.automl.model import SKLearnEstimator +from flaml.automl.task import Task + +_EPS = 1e-7 +# Platt calibration fits n_classes + 1 parameters on the training margins. Past +# this many margin entries it uses a deterministic row subsample, which bounds its +# memory (two float64 arrays of this size) and barely changes the fit. +_CALIBRATION_MAX_ENTRIES = 10_000_000 +# Strength of the pull of the Platt slope (relative to the closed form) towards the +# closed form, against the weighted mean log loss; keeps separable data finite. +_CALIBRATION_PENALTY = 1e-3 +# rows x heads per chunk when scoring the training data +_CHUNK_ENTRIES = 1_000_000 + + +def _class_sums(X, y_index, sample_weight, n_classes): + """Weighted per-class feature sums, shape (n_classes, n_features), in one pass over ``X``.""" + from scipy.sparse import csr_matrix + + onehot = csr_matrix((sample_weight, (y_index, np.arange(X.shape[0]))), shape=(n_classes, X.shape[0])) + sums = onehot @ X + return sums.toarray() if issparse(sums) else np.asarray(sums) + + +def _to_dense_row(values): + """Collapse the result of a column-wise reduction to a 1-D ndarray.""" + if issparse(values): + values = values.toarray() + return np.asarray(values).ravel() + + +def _fit_multinomial(margins, y_index, sample_weight, scale, bias): + """Fit ``softmax(a * margins + b)`` to the labels by weighted maximum likelihood. + + Optimizes ``(a / scale, b)``, starting from the closed form ``(1, bias)``; + ``a`` stays positive. The objective is the weighted mean log loss plus + ``0.5 * _CALIBRATION_PENALTY * (a / scale - 1)^2``. Only the slope can diverge + (on separable data); with it bounded the biases have a finite optimum, and + leaving them unpenalized keeps rare classes' biases free. The objective does + not change when the weights are rescaled (AdaBoost normalizes them to sum to + 1). Each step is + O(n_samples * n_classes), with ``margins`` and one more array of its size in + memory. + """ + from scipy.optimize import minimize + from scipy.special import logsumexp + + rows = np.arange(margins.shape[0]) + total = sample_weight.sum() + start = np.concatenate([[1.0], bias]) + + def loss_and_grad(theta): + logits = margins * (theta[0] * scale) + logits += theta[1:] + norm = logsumexp(logits, axis=1) + stretch = theta[0] - 1.0 + loss = np.dot(sample_weight, norm - logits[rows, y_index]) / total + loss += 0.5 * _CALIBRATION_PENALTY * stretch * stretch + # reuse the buffer for the weighted residuals w * (softmax - onehot) + residual = logits + residual -= norm[:, None] + np.exp(residual, out=residual) + residual[rows, y_index] -= 1.0 + residual *= sample_weight[:, None] + grad = np.empty_like(theta) + grad[0] = scale * np.einsum("ij,ij->", residual, margins) + grad[1:] = residual.sum(axis=0) + grad /= total + grad[0] += _CALIBRATION_PENALTY * stretch + return loss, grad + + bounds = [(1e-8, None)] + [(None, None)] * bias.size + theta = minimize(loss_and_grad, start, jac=True, method="L-BFGS-B", bounds=bounds).x + return float(theta[0] * scale), theta[1:] + + +def _fit_binary(margins, is_positive, sample_weight, scale, bias): + """Fit ``sigmoid(a * margins + b)``: the two-class case of ``_fit_multinomial``, same objective.""" + from scipy.optimize import minimize + from scipy.special import expit + + target = is_positive.astype(np.float64) + total = sample_weight.sum() + + def loss_and_grad(theta): + logits = margins * (theta[0] * scale) + theta[1] + stretch = theta[0] - 1.0 + loss = np.dot(sample_weight, np.logaddexp(0.0, logits) - target * logits) / total + loss += 0.5 * _CALIBRATION_PENALTY * stretch * stretch + residual = sample_weight * (expit(logits) - target) + grad = np.array([scale * np.dot(residual, margins), residual.sum()]) / total + grad[0] += _CALIBRATION_PENALTY * stretch + return loss, grad + + theta = minimize( + loss_and_grad, np.array([1.0, bias]), jac=True, method="L-BFGS-B", bounds=[(1e-8, None), (None, None)] + ).x + return float(theta[0] * scale), np.array([theta[1]]) + + +def _min_value(X): + if issparse(X): + return X.data.min() if X.nnz else 0.0 + return X.min() if X.size else 0.0 + + +class SEFRClassifier(ClassifierMixin, SKLearnBaseEstimator): + """Scalable, Efficient and Fast classifieR (SEFR). + + Binary targets are fit with the closed form of the paper (Eqs. 3-9); + multiclass targets use the one-vs-rest scheme of Sec. 3.4. + + SEFR's weight formula is only meaningful for non-negative features, so every + input reaches the SEFR heads in that domain or is rejected. + + Parameters + ---------- + scaling : {"minmax", "none"}, default="minmax" + "minmax" maps each feature to [0, 1] using the range seen in training, + and clips values outside that range at prediction time. Sparse input is + scaled by the column maximum instead, which preserves sparsity; that is + only a valid [0, 1] map for non-negative data, so sparse input with + negative entries is rejected at fit time. "none" uses the features as + given and requires them to be non-negative. + class_weight : dict, "balanced" or "none", default="none" + Per-class misclassification costs c. They would cancel in the class + means of Eq. 3-4, so they act on the calibrated decision instead: the + closed-form calibration adds log(c) to each class's logit, and "platt" + weights each sample's likelihood by the cost of its class. "balanced" + sets c inversely proportional to each class's total sample weight. + None is accepted as an alias of "none". + threshold : {"sefr", "balanced"}, default="sefr" + "sefr" is the class-count-weighted score average of Eq. 9. "balanced" + is the unweighted midpoint of the two class score means. + threshold_shift : float, default=0.0 + Shifts the bias by this many standard deviations of the training scores. + calibration : {"sigmoid", "platt"}, default="sigmoid" + SEFR produces margins, not probabilities. Margins m map to calibrated + logits a * m + b, with one scale a > 0 shared by all heads and one bias + b per head; probabilities are their sigmoid (binary) or softmax + (multiclass). `decision_function` returns these logits and `predict` + thresholds them at 0 (binary) or takes their argmax (multiclass), so + `predict`, `decision_function` and `predict_proba` always agree. + "sigmoid" is closed form: a is the inverse standard deviation of the + training margins; b is log(c_1 / c_0) for binary targets, which keeps + SEFR's Eq. 9 threshold unless class_weight is set, and the log of the + cost-weighted class prior for multiclass targets. + "platt" fits a and b by weighted maximum likelihood on the training + margins, so the fitted intercept replaces the Eq. 9 threshold. It is + iterative, so it gives up the closed form, but it usually gives a much + better log_loss. + eps : float, default=1e-7 + Stabilizer for the denominator of Eq. 5. + """ + + def __init__( + self, + scaling="minmax", + class_weight="none", + threshold="sefr", + threshold_shift=0.0, + calibration="sigmoid", + eps=_EPS, + ): + self.scaling = scaling + self.class_weight = class_weight + self.threshold = threshold + self.threshold_shift = threshold_shift + self.calibration = calibration + self.eps = eps + + def _more_tags(self): # scikit-learn < 1.6 + return {"requires_positive_X": self.scaling == "none", "X_types": ["2darray", "sparse"]} + + def __sklearn_tags__(self): # scikit-learn >= 1.6 + tags = super().__sklearn_tags__() + tags.input_tags.sparse = True + tags.input_tags.positive_only = self.scaling == "none" + return tags + + def _check_params(self): + if not (self.class_weight in (None, "none", "balanced") or isinstance(self.class_weight, dict)): + raise ValueError(f"class_weight must be a dict, 'balanced' or 'none', got {self.class_weight!r}") + for name, allowed in ( + ("scaling", ("minmax", "none")), + ("threshold", ("sefr", "balanced")), + ("calibration", ("sigmoid", "platt")), + ): + if getattr(self, name) not in allowed: + raise ValueError(f"{name} must be one of {allowed}, got {getattr(self, name)!r}") + for name, positive in (("eps", True), ("threshold_shift", False)): + value = getattr(self, name) + if isinstance(value, bool) or not isinstance(value, numbers.Real) or not np.isfinite(value): + raise ValueError(f"{name} must be a finite number, got {value!r}") + if positive and value <= 0: + raise ValueError(f"{name} must be positive, got {value!r}") + + @staticmethod + def _check_sample_weight(sample_weight, n_samples): + if sample_weight is None: + return np.ones(n_samples, dtype=np.float64) + sample_weight = np.asarray(sample_weight, dtype=np.float64) + if sample_weight.ndim == 0: + sample_weight = np.full(n_samples, float(sample_weight)) + if sample_weight.shape != (n_samples,): + raise ValueError(f"sample_weight.shape == {sample_weight.shape}, expected {(n_samples,)}!") + if not np.all(np.isfinite(sample_weight)): + raise ValueError("sample_weight must be finite") + if np.any(sample_weight < 0): + raise ValueError("sample_weight must be non-negative") + return sample_weight + + def _check_non_negative(self, X, reason): + if _min_value(X) < 0: + raise ValueError( + f"Negative values in data passed to SEFRClassifier: SEFR requires non-negative features {reason}. " + "Use dense input with scaling='minmax', or shift the features to be non-negative." + ) + + def _class_totals(self, y_index, sample_weight): + """Per-class weight totals; a class with none would divide by zero in Eq. 3-4.""" + totals = np.bincount(y_index, weights=sample_weight, minlength=self.classes_.size) + if np.any(totals <= 0): + raise ValueError( + f"Sample weights sum to zero for classes {self.classes_[totals <= 0].tolist()}; " + "every class needs a positive total weight" + ) + return totals + + def _fit_scaler(self, X): + if issparse(X): + self._check_non_negative(X, "for sparse input") + if self.scaling == "none": + self._check_non_negative(X, "when scaling='none'") + self.offset_, self.spread_ = None, None + return + if issparse(X): + # the input is non-negative, so dividing by the column maximum maps + # it into [0, 1]; subtracting a column minimum would densify it + self.offset_ = None + spread = _to_dense_row(X.max(axis=0)) + else: + self.offset_ = X.min(axis=0) + spread = X.max(axis=0) - self.offset_ + spread[spread == 0] = 1.0 + self.spread_ = spread + + def _scale(self, X): + if self.spread_ is None: + self._check_non_negative(X, "when scaling='none'") + return X + if issparse(X) and self.offset_ is not None and np.any(self.offset_): + # fitted on dense data with a nonzero minimum: subtracting it densifies + # the input anyway, and skipping it would score sparse input differently + X = X.toarray() + if issparse(X): + # multiply() keeps the matrix sparse; dividing by a dense row would not + X = X.multiply(1.0 / self.spread_).tocsr() + np.clip(X.data, 0.0, 1.0, out=X.data) + return X + if self.offset_ is not None: + X = X - self.offset_ + return np.clip(X / self.spread_, 0.0, 1.0) + + def _class_costs(self, class_totals): + if self.class_weight == "balanced": + return class_totals.sum() / (self.classes_.size * class_totals) + if isinstance(self.class_weight, dict): + costs = np.array([self.class_weight.get(label, 1.0) for label in self.classes_], dtype=np.float64) + if not np.all(np.isfinite(costs)) or np.any(costs < 0): + raise ValueError("class_weight values must be finite and non-negative") + return costs + return np.ones(self.classes_.size) + + def _fit_heads(self, X, y_index, sample_weight, class_totals): + """Closed-form coef and Eq. 9 bias of every one-vs-rest head, from one pass over ``X``.""" + sums = _class_sums(X, y_index, sample_weight, self.classes_.size) + if self.classes_.size == 2: + pos_sums, pos_totals = sums[[1]], class_totals[[1]] + neg_sums, neg_totals = sums[[0]], class_totals[[0]] + else: + pos_sums, pos_totals = sums, class_totals + neg_sums, neg_totals = sums.sum(axis=0) - sums, class_totals.sum() - class_totals + avg_pos = pos_sums / pos_totals[:, None] + avg_neg = neg_sums / neg_totals[:, None] + coef = (avg_pos - avg_neg) / (avg_pos + avg_neg + self.eps) + + # scores are linear in x, so each class's mean score is its mean row times coef + score_pos = np.einsum("hd,hd->h", avg_pos, coef) + score_neg = np.einsum("hd,hd->h", avg_neg, coef) + if self.threshold == "balanced": + bias = 0.5 * (score_pos + score_neg) + else: + bias = (neg_totals * score_pos + pos_totals * score_neg) / (neg_totals + pos_totals) + score_mean = (sums.sum(axis=0) / class_totals.sum()) @ coef.T + return coef, bias, score_mean + + def _score_std(self, X, sample_weight, score_mean): + """Weighted standard deviation of each head's training scores, in row chunks.""" + chunk = max(1, _CHUNK_ENTRIES // self.coef_.shape[0]) + second = np.zeros(self.coef_.shape[0]) + for start in range(0, X.shape[0], chunk): + centered = np.asarray(X[start : start + chunk] @ self.coef_.T) - score_mean + second += sample_weight[start : start + chunk] @ (centered * centered) + return np.sqrt(second / sample_weight.sum()) + + def _calibration_rows(self, n_samples, y_index, weights): + """All rows, or a deterministic subsample that keeps a positive-weight row of every class.""" + n_logits = 2 if self.classes_.size == 2 else self.classes_.size + if n_samples * n_logits <= _CALIBRATION_MAX_ENTRIES: + return np.arange(n_samples) + stride = -(-n_samples * n_logits // _CALIBRATION_MAX_ENTRIES) + positive = np.flatnonzero(weights > 0) + first_positive = positive[np.unique(y_index[positive], return_index=True)[1]] + rows = np.union1d(np.arange(0, n_samples, stride), first_positive) + # a class without weight in the subsample would drive its bias to -inf + self._class_totals(y_index[rows], weights[rows]) + return rows + + def _fit_calibration(self, X, y_index, sample_weight, costs, score_mean, score_std): + """Return the scale ``a`` and per-head biases ``b`` of the calibrated logits.""" + # pooled spread of the margins (scores - intercept) over every head and sample + margin_mean = score_mean - self.intercept_ + spread = np.sqrt(np.mean(score_std**2 + (margin_mean - margin_mean.mean()) ** 2)) + scale = 1.0 / (spread or 1.0) + weights = sample_weight * costs[y_index] + if self.classes_.size == 2: + bias = np.log(costs[[1]] / costs[[0]]) + else: + bias = np.log(np.bincount(y_index, weights=weights, minlength=self.classes_.size) / weights.sum()) + if self.calibration == "sigmoid": + return scale, bias + + rows = self._calibration_rows(X.shape[0], y_index, weights) + margins = np.asarray(X[rows] @ self.coef_.T) - self.intercept_ + if self.classes_.size == 2: + return _fit_binary(margins[:, 0], y_index[rows] == 1, weights[rows], scale, bias[0]) + return _fit_multinomial(margins, y_index[rows], weights[rows], scale, bias) + + def fit(self, X, y, sample_weight=None): + self._check_params() + X, y = _validate_data(self, X=X, y=y, accept_sparse="csr", dtype=np.float64) + check_classification_targets(y) + self.classes_, y_index = np.unique(y, return_inverse=True) + if self.classes_.size < 2: + raise ValueError( + "SEFR needs samples of at least 2 classes in the data, " + f"but the data contains only one class: {self.classes_[0]}" + ) + + sample_weight = self._check_sample_weight(sample_weight, X.shape[0]) + class_totals = self._class_totals(y_index, sample_weight) + costs = self._class_costs(class_totals) + self._class_totals(y_index, sample_weight * costs[y_index]) + + # zero-weight samples take no part in the fit, including the feature range + self._fit_scaler(X if np.all(sample_weight > 0) else X[sample_weight > 0]) + X = self._scale(X) + + self.coef_, self.intercept_, score_mean = self._fit_heads(X, y_index, sample_weight, class_totals) + score_std = self._score_std(X, sample_weight, score_mean) + if self.threshold_shift: + self.intercept_ = self.intercept_ + self.threshold_shift * np.where(score_std > 0, score_std, 1.0) + self.calibration_scale_, self.calibration_bias_ = self._fit_calibration( + X, y_index, sample_weight, costs, score_mean, score_std + ) + return self + + def decision_function(self, X): + check_is_fitted(self) + X = _validate_data(self, X=X, accept_sparse="csr", dtype=np.float64, reset=False) + scores = np.asarray(self._scale(X) @ self.coef_.T) - self.intercept_ + scores = scores * self.calibration_scale_ + self.calibration_bias_ + return scores.ravel() if self.coef_.shape[0] == 1 else scores + + def predict(self, X): + scores = self.decision_function(X) + if self.coef_.shape[0] == 1: + return self.classes_[(scores > 0).astype(int)] + return self.classes_[np.argmax(scores, axis=1)] + + def predict_proba(self, X): + scores = self.decision_function(X) + if self.coef_.shape[0] == 1: + proba = 1.0 / (1.0 + np.exp(-np.clip(scores, -30, 30))) + return np.column_stack([1.0 - proba, proba]) + proba = np.exp(scores - scores.max(axis=1, keepdims=True)) + return proba / proba.sum(axis=1, keepdims=True) + + +def _adaboost_estimator_param(): + """AdaBoostClassifier's ``base_estimator`` was renamed ``estimator`` in scikit-learn 1.2.""" + params = inspect.signature(AdaBoostClassifier.__init__).parameters + return "estimator" if "estimator" in params else "base_estimator" + + +class SEFREstimator(SKLearnEstimator): + """The class for tuning SEFR.""" + + @classmethod + def search_space(cls, **params) -> dict: + return { + "class_weight": { + "domain": tune.choice(["none", "balanced"]), + "init_value": "none", + }, + "threshold": { + "domain": tune.choice(["sefr", "balanced"]), + "init_value": "sefr", + }, + "threshold_shift": { + "domain": tune.uniform(lower=-1.0, upper=1.0), + "init_value": 0.0, + }, + "calibration": { + "domain": tune.choice(["sigmoid", "platt"]), + "init_value": "sigmoid", + }, + } + + @classmethod + def cost_relative2lgbm(cls) -> float: + return 1.0 + + def config2params(self, config: dict) -> dict: + params = super().config2params(config) + params.pop("n_jobs", None) + params.pop("random_state", None) + return params + + def __init__(self, task: Task = "binary", **config): + super().__init__(task, **config) + assert self._task.is_classification(), "SEFR is for classification tasks only" + self.estimator_class = SEFRClassifier + + +class SEFRBoostEstimator(SKLearnEstimator): + """The class for tuning AdaBoost over SEFR base learners. + + SEFR has no hyperparameters of its own to trade accuracy against cost, so on + its own it gives the search very little to do. Boosting it keeps the + closed-form base learner while giving FLAML a real, cheap-to-traverse search + space. + """ + + ITER_HP = "n_estimators" + DEFAULT_ITER = 50 + + @classmethod + def search_space(cls, data_size, **params) -> dict: + upper = max(5, min(1024, int(data_size[0]))) + space = { + "n_estimators": { + "domain": tune.lograndint(lower=4, upper=upper), + "init_value": 4, + "low_cost_init_value": 4, + }, + "learning_rate": { + "domain": tune.loguniform(lower=1 / 1024, upper=1.0), + "init_value": 0.1, + }, + } + base_space = SEFREstimator.search_space(**params) + space["class_weight"] = base_space["class_weight"] + # AdaBoost's SAMME.R weights the base learners by their probabilities, + # so fitted calibration pays off more here than for a single SEFR model + space["calibration"] = dict(base_space["calibration"], init_value="platt") + return space + + @classmethod + def cost_relative2lgbm(cls) -> float: + return 5.0 + + def config2params(self, config: dict) -> dict: + params = super().config2params(config) + params.pop("n_jobs", None) + base_params = {key: params.pop(key) for key in ("scaling", "class_weight", "calibration") if key in params} + params[_adaboost_estimator_param()] = SEFRClassifier(**base_params) + if "random_state" not in params: + params["random_state"] = 24092023 + return params + + def __init__(self, task: Task = "binary", **config): + super().__init__(task, **config) + assert self._task.is_classification(), "SEFR is for classification tasks only" + self.estimator_class = AdaBoostClassifier diff --git a/flaml/automl/task/generic_task.py b/flaml/automl/task/generic_task.py index 2efeaf8610..89f0c67143 100644 --- a/flaml/automl/task/generic_task.py +++ b/flaml/automl/task/generic_task.py @@ -46,6 +46,7 @@ def estimators(self): if self._estimators is None: # put this into a function to avoid circular dependency from flaml.automl.contrib.histgb import HistGradientBoostingEstimator + from flaml.automl.contrib.sefr import SEFRBoostEstimator, SEFREstimator from flaml.automl.model import ( CatBoostEstimator, ElasticNetEstimator, @@ -87,6 +88,8 @@ def estimators(self): "transformer": TransformersEstimator, "transformer_ms": TransformersEstimatorModelSelection, "histgb": HistGradientBoostingEstimator, + "sefr": SEFREstimator, + "sefr_boost": SEFRBoostEstimator, "svc": SVCEstimator, "sgd": SGDEstimator, "nb_spark": SparkNaiveBayesEstimator, diff --git a/test/automl/test_extra_models.py b/test/automl/test_extra_models.py index 59431fec2b..b0be6c90a0 100644 --- a/test/automl/test_extra_models.py +++ b/test/automl/test_extra_models.py @@ -339,6 +339,12 @@ def test_lassolars(self): _test_regular_models("lassolars", "regression") _test_forecast("lassolars") + def test_sefr(self): + _test_regular_models("sefr", "classification") + + def test_sefr_boost(self): + _test_regular_models("sefr_boost", "classification") + def test_seasonal_naive(self): _test_forecast("snaive") diff --git a/test/automl/test_model.py b/test/automl/test_model.py index 78e3c6a706..319e9d6986 100644 --- a/test/automl/test_model.py +++ b/test/automl/test_model.py @@ -1,13 +1,18 @@ +import inspect import platform import sys from datetime import datetime import numpy as np import pytest +import scipy.sparse from pandas import DataFrame from sklearn.datasets import make_classification +from sklearn.metrics import log_loss +from sklearn.utils.estimator_checks import check_estimator from flaml.automl.contrib.histgb import HistGradientBoostingEstimator +from flaml.automl.contrib.sefr import SEFRBoostEstimator, SEFRClassifier, SEFREstimator from flaml.automl.model import ( BaseEstimator, CatBoostEstimator, @@ -148,6 +153,263 @@ def test_prep(): print(lgbm.feature_importances_) +def test_sefr_matches_closed_form(): + """SEFR is a closed form; check the fitted model against it directly.""" + X, y = make_classification(200, 8, random_state=0) + X = np.abs(X) + clf = SEFRClassifier(scaling="none").fit(X, y) + + avg_pos = X[y == 1].mean(axis=0) + avg_neg = X[y == 0].mean(axis=0) + expected_coef = (avg_pos - avg_neg) / (avg_pos + avg_neg + 1e-7) + scores = X @ expected_coef + n_pos, n_neg = int((y == 1).sum()), int((y == 0).sum()) + expected_bias = (n_neg * scores[y == 1].mean() + n_pos * scores[y == 0].mean()) / (n_pos + n_neg) + + np.testing.assert_allclose(clf.coef_.ravel(), expected_coef) + np.testing.assert_allclose(clf.intercept_[0], expected_bias) + # a binary head is n_features + 1 floats + assert clf.coef_.size + clf.intercept_.size == 9 + + +def test_sefr_multiclass_matches_one_vs_rest(): + """The one-pass multiclass fit equals fitting each one-vs-rest head separately.""" + X, y = make_classification(300, 8, n_classes=4, n_informative=6, random_state=0) + X = np.abs(X) + w = np.random.RandomState(0).uniform(0.1, 3.0, len(y)) + clf = SEFRClassifier(scaling="none", threshold_shift=0.3).fit(X, y, sample_weight=w) + for k in range(4): + pos, neg = y == k, y != k + avg_pos = np.average(X[pos], axis=0, weights=w[pos]) + avg_neg = np.average(X[neg], axis=0, weights=w[neg]) + coef = (avg_pos - avg_neg) / (avg_pos + avg_neg + 1e-7) + scores = X @ coef + w_pos, w_neg = w[pos].sum(), w[neg].sum() + bias = (w_neg * np.average(scores[pos], weights=w[pos]) + w_pos * np.average(scores[neg], weights=w[neg])) / ( + w_pos + w_neg + ) + std = np.sqrt(np.average((scores - np.average(scores, weights=w)) ** 2, weights=w)) + np.testing.assert_allclose(clf.coef_[k], coef) + np.testing.assert_allclose(clf.intercept_[k], bias + 0.3 * std) + + +def test_sefr_sample_weight(): + X, y = make_classification(200, 8, random_state=0) + for calibration in ("sigmoid", "platt"): + base = SEFRClassifier(calibration=calibration).fit(X, y) + uniform = SEFRClassifier(calibration=calibration).fit(X, y, sample_weight=np.full(len(y), 3.0)) + # a constant weight rescales both class means identically, so it is a no-op + np.testing.assert_allclose(base.coef_, uniform.coef_) + np.testing.assert_allclose(base.intercept_, uniform.intercept_) + + weights = np.random.RandomState(0).uniform(0.1, 5.0, len(y)) + weighted = SEFRClassifier(calibration=calibration).fit(X, y, sample_weight=weights) + assert not np.allclose(base.coef_, weighted.coef_) + # the weights reach the calibration too, not only the SEFR head + assert base.calibration_scale_ != weighted.calibration_scale_ + + # a zero weight is the same as dropping the sample, feature range included + weights[:50] = 0 + zeroed = SEFRClassifier(calibration=calibration).fit(X, y, sample_weight=weights) + dropped = SEFRClassifier(calibration=calibration).fit(X[50:], y[50:], sample_weight=weights[50:]) + np.testing.assert_allclose(zeroed.coef_, dropped.coef_) + np.testing.assert_allclose(zeroed.calibration_scale_, dropped.calibration_scale_, rtol=1e-5) + + +def test_sefr_invalid_sample_weight(): + X, y = make_classification(200, 8, random_state=0) + weights = np.ones(len(y)) + weights[y == 0] = 0 + with pytest.raises(ValueError, match="sum to zero"): + SEFRClassifier(class_weight="balanced").fit(X, y, sample_weight=weights) + with pytest.raises(ValueError, match="non-negative"): + SEFRClassifier().fit(X, y, sample_weight=-np.ones(len(y))) + with pytest.raises(ValueError, match="finite"): + SEFRClassifier().fit(X, y, sample_weight=np.full(len(y), np.nan)) + + +def test_sefr_invalid_params(): + X, y = make_classification(200, 8, random_state=0) + for eps in (0, -1e-7, np.nan, np.inf, "1e-7", True): + with pytest.raises(ValueError, match="eps must be"): + SEFRClassifier(eps=eps).fit(X, y) + for shift in (np.nan, np.inf, -np.inf, "0.5", None): + with pytest.raises(ValueError, match="threshold_shift must be"): + SEFRClassifier(threshold_shift=shift).fit(X, y) + # numpy scalars, which FLAML's search passes, are accepted + SEFRClassifier(eps=np.float64(1e-6), threshold_shift=np.float64(-0.3)).fit(X, y) + + +def test_sefr_multiclass_and_proba(): + X, y = make_classification(300, 8, n_classes=3, n_informative=5, random_state=0) + for calibration in ("sigmoid", "platt"): + clf = SEFRClassifier(calibration=calibration).fit(X, y) + assert clf.coef_.shape == (3, 8) + proba = clf.predict_proba(X) + assert proba.shape == (300, 3) + np.testing.assert_allclose(proba.sum(axis=1), 1.0) + np.testing.assert_array_equal(clf.classes_[proba.argmax(axis=1)], clf.predict(X)) + np.testing.assert_array_equal(clf.decision_function(X).argmax(axis=1), proba.argmax(axis=1)) + + +def test_sefr_multiclass_calibration(): + X, y = make_classification(600, 8, n_classes=4, n_informative=6, weights=[0.55, 0.25, 0.15], random_state=0) + closed = SEFRClassifier().fit(X, y) + fitted = SEFRClassifier(calibration="platt").fit(X, y) + # the closed form uses the log class prior as the per-class bias + prior = np.bincount(y) / len(y) + np.testing.assert_allclose(closed.calibration_bias_, np.log(prior)) + # maximum likelihood starts from the closed form, so it can only improve on it + assert log_loss(y, fitted.predict_proba(X)) < log_loss(y, closed.predict_proba(X)) + assert fitted.calibration_scale_ > 0 + + # sample weights reach the fitted biases + weights = np.where(y == 0, 5.0, 1.0) + weighted = SEFRClassifier(calibration="platt").fit(X, y, sample_weight=weights) + assert weighted.calibration_bias_[0] - weighted.calibration_bias_[1] > ( + fitted.calibration_bias_[0] - fitted.calibration_bias_[1] + ) + + +def test_sefr_binary_platt_intercept(): + """Binary Platt fits a slope and an intercept, like a (nearly unregularized) logistic regression.""" + from sklearn.linear_model import LogisticRegression + + # noisy labels keep the data far from separable, where the weak penalty is negligible + X, y = make_classification(5000, 8, weights=[0.8], flip_y=0.2, class_sep=0.5, random_state=0) + w = np.random.RandomState(0).uniform(0.1, 3.0, len(y)) + closed = SEFRClassifier().fit(X, y, sample_weight=w) + fitted = SEFRClassifier(calibration="platt").fit(X, y, sample_weight=w) + margins = (closed.decision_function(X) / closed.calibration_scale_).reshape(-1, 1) + reference = LogisticRegression(C=1e12, tol=1e-10, max_iter=10000).fit(margins, y, sample_weight=w) + # the only difference is the weak pull of the slope towards the closed form + np.testing.assert_allclose(fitted.calibration_scale_, reference.coef_[0, 0], rtol=5e-3) + np.testing.assert_allclose(fitted.calibration_bias_[0], reference.intercept_[0], rtol=5e-3, atol=1e-3) + assert fitted.calibration_bias_[0] != 0 + # rescaling the weights (AdaBoost normalizes them) does not change the calibration + rescaled = SEFRClassifier(calibration="platt").fit(X, y, sample_weight=w / w.sum()) + np.testing.assert_allclose(rescaled.calibration_scale_, fitted.calibration_scale_, rtol=1e-6) + np.testing.assert_allclose(rescaled.calibration_bias_, fitted.calibration_bias_, rtol=1e-6, atol=1e-8) + # predict follows the calibrated score, which includes the intercept + np.testing.assert_array_equal(fitted.predict(X), fitted.classes_[(fitted.decision_function(X) > 0).astype(int)]) + + +def test_sefr_class_weight(): + X, y = make_classification(600, 8, weights=[0.8], random_state=0) + base = SEFRClassifier().fit(X, y) + costly = SEFRClassifier(class_weight={0: 1.0, 1: 4.0}).fit(X, y) + # class weights cancel in the class means, and act as costs on the calibrated logit + np.testing.assert_allclose(costly.coef_, base.coef_) + np.testing.assert_allclose(costly.decision_function(X), base.decision_function(X) + np.log(4.0)) + assert (costly.predict(X) == 1).sum() > (base.predict(X) == 1).sum() + + # "balanced" favours the minority class, with either calibration + for calibration in ("sigmoid", "platt"): + plain = SEFRClassifier(calibration=calibration).fit(X, y) + balanced = SEFRClassifier(calibration=calibration, class_weight="balanced").fit(X, y) + recall = lambda clf: (clf.predict(X)[y == 1] == 1).mean() # noqa: E731 + assert recall(balanced) > recall(plain) + + # multiclass: the closed-form bias is the log of the cost-weighted prior + X, y = make_classification(600, 8, n_classes=3, n_informative=6, random_state=0) + costs = {0: 2.0, 1: 1.0, 2: 0.5} + clf = SEFRClassifier(class_weight=costs).fit(X, y) + weighted = np.bincount(y) * np.array([2.0, 1.0, 0.5]) + np.testing.assert_allclose(clf.calibration_bias_, np.log(weighted / weighted.sum())) + with pytest.raises(ValueError, match="sum to zero"): + SEFRClassifier(class_weight={0: 0.0}).fit(X, y) + + +def test_sefr_calibration_subsample(monkeypatch): + import flaml.automl.contrib.sefr as sefr + + X, y = make_classification(2000, 8, n_classes=4, n_informative=6, random_state=0) + full = SEFRClassifier(calibration="platt").fit(X, y) + monkeypatch.setattr(sefr, "_CALIBRATION_MAX_ENTRIES", 1000) + sub = SEFRClassifier(calibration="platt").fit(X, y) + np.testing.assert_allclose(full.coef_, sub.coef_) + assert np.all(np.isfinite(sub.calibration_bias_)) + assert abs(log_loss(y, sub.predict_proba(X)) - log_loss(y, full.predict_proba(X))) < 0.05 + + # rows on the subsample grid, and the first row of every class, all weigh zero: + # the subsample must still reach a positive-weight row of every class + rows = np.arange(len(y)) + weights = np.ones(len(y)) + weights[rows % 8 == 0] = 0 + weights[[np.flatnonzero(y == k)[0] for k in range(4)]] = 0 + zeroed = SEFRClassifier(calibration="platt").fit(X, y, sample_weight=weights) + assert np.all(np.isfinite(zeroed.calibration_bias_)) + picked = zeroed._calibration_rows(len(y), y, weights) + assert all(weights[picked][y[picked] == k].sum() > 0 for k in range(4)) + + +def test_sefr_sparse(): + X, y = make_classification(200, 8, random_state=0) + X = np.where(np.abs(X) > 0.5, np.abs(X), 0.0) + # every column holds a zero, so dense min-max scaling and sparse max scaling agree + dense = SEFRClassifier().fit(X, y) + sparse = SEFRClassifier().fit(scipy.sparse.csr_matrix(X), y) + np.testing.assert_allclose(dense.coef_, sparse.coef_) + np.testing.assert_allclose(dense.decision_function(X), sparse.decision_function(scipy.sparse.csr_matrix(X))) + + # a model fitted on dense data scores dense and sparse input identically, both + # when its fitted minimum is zero and when it is not + for data in (X, X + 1.0): + fitted = SEFRClassifier().fit(data, y) + for probe in (data, data - 0.5, data * 2): + np.testing.assert_allclose( + fitted.decision_function(scipy.sparse.csr_matrix(probe)), fitted.decision_function(probe) + ) + + # sparse input cannot be shifted into [0, 1] without densifying, so negatives are rejected + signed = scipy.sparse.csr_matrix(np.where(y[:, None] == 1, X, -X)) + with pytest.raises(ValueError, match="Negative values"): + SEFRClassifier().fit(signed, y) + with pytest.raises(ValueError, match="Negative values"): + SEFRClassifier(scaling="none").fit(-X, y) + # out-of-range values at prediction time are clipped into [0, 1] + assert np.all(sparse._scale(-signed).data >= 0) + assert np.all(dense._scale(X * 3 - 1) >= 0) and np.all(dense._scale(X * 3 - 1) <= 1) + + +def test_sefr_sklearn_estimator_checks(): + for estimator in (SEFRClassifier(), SEFRClassifier(calibration="platt")): + if "on_fail" in inspect.signature(check_estimator).parameters: # scikit-learn >= 1.6 + results = check_estimator(estimator, on_fail=None) + failed = [r["check_name"] for r in results if r["status"] == "failed"] + else: + failed = [] + for est, check in check_estimator(estimator, generate_only=True): + try: + check(est) + except Exception: + failed.append(getattr(check, "func", check).__name__) + assert not failed, failed + # sparse support is declared to scikit-learn < 1.6 too + assert "sparse" in SEFRClassifier()._more_tags()["X_types"] + + +def test_sefr_estimators(): + X, y = make_classification(200, 8, random_state=0) + assert set(SEFREstimator.search_space()) == { + "class_weight", + "threshold", + "threshold_shift", + "calibration", + } + assert "n_estimators" in SEFRBoostEstimator.search_space(data_size=(200, 8)) + + sefr = SEFREstimator(task="binary") + sefr.fit(X, y) + assert sefr.predict(X).shape == (200,) + assert sefr.predict_proba(X).shape == (200, 2) + assert sefr.feature_importances_ is not None + + boost = SEFRBoostEstimator(task="binary", n_estimators=4) + boost.fit(X, y) + assert boost.predict_proba(X).shape == (200, 2) + + if __name__ == "__main__": test_lrl2() test_prep() diff --git a/website/docs/Use-Cases/Task-Oriented-AutoML.md b/website/docs/Use-Cases/Task-Oriented-AutoML.md index 3c377ccf15..288f473b7e 100644 --- a/website/docs/Use-Cases/Task-Oriented-AutoML.md +++ b/website/docs/Use-Cases/Task-Oriented-AutoML.md @@ -185,6 +185,8 @@ The estimator list can contain one or more estimator names, each corresponding t - 'extra_tree': ExtraTreesEstimator for task "classification", "regression", "ts_forecast" and "ts_forecast_classification". Hyperparameters: n_estimators, max_features, max_leaves, criterion (for classification only). Starting from v1.1.0, it uses a fixed random_state by default. - 'histgb': HistGradientBoostingEstimator for task "classification", "regression", "ts_forecast" and "ts_forecast_classification". Hyperparameters: n_estimators, max_leaves, min_samples_leaf, learning_rate, log_max_bin (logarithm of (max_bin + 1) with base 2), l2_regularization. It uses a fixed random_state by default. + - 'sefr': SEFREstimator for task "classification". A linear-time, closed-form classifier ([SEFR](https://arxiv.org/abs/2006.04620)): fitting takes three passes over the data whatever the number of classes, with no iterative optimization, so it is one of the cheapest estimators in the portfolio and likely to return a model under a very small `time_budget`. Hyperparameters: class_weight (misclassification costs applied to the calibrated decision), threshold, threshold_shift, calibration ('sigmoid' is closed form; 'platt' fits the probability scale and a bias per class iteratively, on a row subsample for very large data). It min-max scales features internally; sparse input must be non-negative. + - 'sefr_boost': SEFRBoostEstimator for task "classification". AdaBoost over SEFR base learners. Hyperparameters: n_estimators, learning_rate, class_weight, calibration. It uses a fixed random_state by default. - 'lrl1': LRL1Classifier (sklearn.LogisticRegression with L1 regularization) for task "classification". Hyperparameters: C. - 'lrl2': LRL2Classifier (sklearn.LogisticRegression with L2 regularization) for task "classification". Hyperparameters: C. - 'catboost': CatBoostEstimator for task "classification" and "regression". Hyperparameters: early_stopping_rounds, learning_rate, n_estimators.