From daf21f537438f83de531876d0dd9c12e5dc64c48 Mon Sep 17 00:00:00 2001 From: MichalRedm <45372892+MichalRedm@users.noreply.github.com> Date: Fri, 2 Oct 2026 12:48:48 +0200 Subject: [PATCH 1/3] refactor(api): decouple optional dependencies and promote river to core - Promote river to core dependencies in pyproject.toml as foundational streaming drift engine - Enhance OptionalDependencyError to produce clear, actionable pip install extra suggestions - Guard optional imports (umap-learn, shap, lime, pyclustering, hdbscan, tensorflow, matplotlib, seaborn, click) across core modules - Add lazy import wrappers raising OptionalDependencyError on actual usage --- pyproject.toml | 1 + src/stride/common/__init__.py | 11 ++++- src/stride/datasets/protree_data/__init__.py | 43 +++++++++++-------- src/stride/exceptions.py | 26 ++++++++--- src/stride/plotting/_renderers.py | 8 +++- src/stride/plotting/stream.py | 28 +++++++++++- src/stride/xai/boundary/analysis.py | 16 ++++--- src/stride/xai/boundary/ssnp.py | 28 +++++++++--- src/stride/xai/clustering/__init__.py | 16 ++++--- src/stride/xai/clustering/xmeans.py | 11 ++++- src/stride/xai/importance/__init__.py | 12 ++++-- src/stride/xai/importance/methods.py | 21 ++++++++- src/stride/xai/recurrence/methods.py | 16 ++++--- .../xai/recurrence/protree/metrics/compare.py | 2 - src/stride/xai/stats/__init__.py | 6 ++- 15 files changed, 181 insertions(+), 64 deletions(-) diff --git a/pyproject.toml b/pyproject.toml index cb82df8..a6f3d57 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -30,6 +30,7 @@ dependencies = [ "pandas", "scipy", "scikit-learn", + "river", ] [project.optional-dependencies] diff --git a/src/stride/common/__init__.py b/src/stride/common/__init__.py index e217222..0feb7ac 100644 --- a/src/stride/common/__init__.py +++ b/src/stride/common/__init__.py @@ -9,7 +9,8 @@ from sklearn.discriminant_analysis import LinearDiscriminantAnalysis as LDA from sklearn.manifold import MDS, TSNE, LocallyLinearEmbedding from sklearn.preprocessing import MinMaxScaler, StandardScaler -from umap import UMAP + +from stride.exceptions import OptionalDependencyError class ScalingType(Enum): @@ -103,6 +104,14 @@ def _create_reducer(self) -> Any: elif self.reducer_type == ReducerType.TSNE: return TSNE(n_components=self.n_components, init="pca", learning_rate="auto", random_state=42) elif self.reducer_type == ReducerType.UMAP: + try: + from umap import UMAP + except ImportError as err: + raise OptionalDependencyError( + package_name="umap-learn", + feature_name="UMAP dimensionality reduction", + extra_name="clustering", + ) from err return UMAP(n_components=self.n_components, random_state=42, transform_seed=42) elif self.reducer_type == ReducerType.LLE: return LocallyLinearEmbedding(n_components=self.n_components, n_neighbors=max(5, self.n_components + 1)) diff --git a/src/stride/datasets/protree_data/__init__.py b/src/stride/datasets/protree_data/__init__.py index 2cdae75..73399fb 100644 --- a/src/stride/datasets/protree_data/__init__.py +++ b/src/stride/datasets/protree_data/__init__.py @@ -1,25 +1,32 @@ -from __future__ import annotations - -import click +try: + import click +except ImportError: + click = None from stride.datasets.protree_data.static import download_all, DEFAULT_DATA_DIR -@click.command() -@click.option("--directory", "-d", default=DEFAULT_DATA_DIR, help="Directory to store datasets") -@click.option("--silent", "-s", is_flag=True, help="Suppress displaying progress.") -@click.option( - "--dataset-names", - "-n", - default="all", - help="Comma-separated list of dataset names to download. " - "Allowable values are 'breast_cancer', 'caltech', 'compass', " - "'diabetes', 'mnist' and 'rhc'. Use 'all' to download all " - "datasets.", -) -def main(directory, silent, dataset_names): - download_all(directory=directory, dataset_names=[s.strip() for s in dataset_names.split(",")], verbose=not silent) +def _cli_entry(): + if click is None: + raise ImportError("Dataset download CLI requires 'click'. Install with: pip install click") + + @click.command() + @click.option("--directory", "-d", default=DEFAULT_DATA_DIR, help="Directory to store datasets") + @click.option("--silent", "-s", is_flag=True, help="Suppress displaying progress.") + @click.option( + "--dataset-names", + "-n", + default="all", + help="Comma-separated list of dataset names to download. " + "Allowable values are 'breast_cancer', 'caltech', 'compass', " + "'diabetes', 'mnist' and 'rhc'. Use 'all' to download all " + "datasets.", + ) + def main(directory, silent, dataset_names): + download_all(directory=directory, dataset_names=[s.strip() for s in dataset_names.split(",")], verbose=not silent) + + return main() if __name__ == "__main__": - main() + _cli_entry() diff --git a/src/stride/exceptions.py b/src/stride/exceptions.py index 759ac39..ff7c746 100644 --- a/src/stride/exceptions.py +++ b/src/stride/exceptions.py @@ -8,13 +8,25 @@ class StrideError(Exception): class OptionalDependencyError(StrideError): """Raised when an optional dependency (e.g. tensorflow, shap) is missing.""" - def __init__(self, package_name: str, feature_name: str): - super().__init__( - f"Feature '{feature_name}' requires optional dependency '{package_name}'. " - f"Install it using: pip install stride-xai[{package_name}] or pip install {package_name}" - ) - self.package_name = package_name - self.feature_name = feature_name + def __init__( + self, + package_name: str, + feature_name: str | None = None, + extra_name: str | None = None, + ): + if feature_name is None: + super().__init__(package_name) + self.package_name = package_name + self.feature_name = "" + self.extra_name = "" + else: + self.package_name = package_name + self.feature_name = feature_name + self.extra_name = extra_name or package_name + super().__init__( + f"Feature '{feature_name}' requires optional dependency '{package_name}'. " + f"Install it using: pip install stride-xai[{self.extra_name}] or pip install {package_name}" + ) class DriftDetectionError(StrideError): diff --git a/src/stride/plotting/_renderers.py b/src/stride/plotting/_renderers.py index 9a1182f..9992cb9 100644 --- a/src/stride/plotting/_renderers.py +++ b/src/stride/plotting/_renderers.py @@ -1,3 +1,5 @@ +from __future__ import annotations + """Low-level matplotlib rendering primitives for stream visualisation. These are private helpers used internally by :mod:`.stream`. They are not @@ -5,7 +7,11 @@ """ import numpy as np -import matplotlib.pyplot as plt + +try: + import matplotlib.pyplot as plt +except ImportError: + plt = None def _plot_violin(ax: plt.Axes, values: list, positions: list, colors: list, alphas: list) -> None: diff --git a/src/stride/plotting/stream.py b/src/stride/plotting/stream.py index f8f92a8..ee191a0 100644 --- a/src/stride/plotting/stream.py +++ b/src/stride/plotting/stream.py @@ -7,15 +7,35 @@ - :func:`visualize_data_stream` """ +from typing import Any import numpy as np -import matplotlib.pyplot as plt import pandas as pd -from matplotlib.figure import Figure from sklearn.decomposition import PCA +from stride.exceptions import OptionalDependencyError + +try: + import matplotlib.pyplot as plt + from matplotlib.figure import Figure + + _HAS_MATPLOTLIB = True +except ImportError: + plt = None + Figure = Any # type: ignore + _HAS_MATPLOTLIB = False + from ._renderers import _plot_distribution_comparison +def _ensure_matplotlib() -> None: + if not _HAS_MATPLOTLIB: + raise OptionalDependencyError( + package_name="matplotlib", + feature_name="Stream plotting", + extra_name="vis", + ) + + def plot_feature_target_relationship( X, n_features, @@ -60,6 +80,7 @@ def plot_feature_target_relationship( ------- matplotlib.figure.Figure """ + _ensure_matplotlib() unique_classes = sorted(np.unique(np.concatenate([y_before, y_after]))) n_classes = len(unique_classes) @@ -122,6 +143,7 @@ def plot_class_distribution(class_dist_before, class_dist_after, class_colors, t ------- matplotlib.figure.Figure """ + _ensure_matplotlib() fig, (ax_before, ax_after) = plt.subplots(1, 2, figsize=(12, 6)) if title: fig.suptitle(title, fontsize=16, fontweight="bold", y=1.0) @@ -175,6 +197,7 @@ def plot_feature_space( ------- matplotlib.figure.Figure """ + _ensure_matplotlib() fig, (ax_before, ax_after) = plt.subplots(1, 2, figsize=(14, 7)) fs_title_suffix = "" @@ -307,6 +330,7 @@ def visualize_data_stream( list[matplotlib.figure.Figure] List of three figures in the order described above. """ + _ensure_matplotlib() if isinstance(X, pd.DataFrame): X = X.values if isinstance(y, pd.Series): diff --git a/src/stride/xai/boundary/analysis.py b/src/stride/xai/boundary/analysis.py index 25dd1b3..06c088c 100644 --- a/src/stride/xai/boundary/analysis.py +++ b/src/stride/xai/boundary/analysis.py @@ -1,6 +1,7 @@ import numpy as np import random from sklearn.preprocessing import MinMaxScaler +from stride.exceptions import OptionalDependencyError, StrideError from stride.xai.boundary.disagreement import compute_disagreement_analysis @@ -64,14 +65,15 @@ def analyze(self, model_class=None, model_params=None, grid_size=300, ssnp_epoch else: try: from stride.xai.boundary.ssnp import SSNP - except ImportError as err: - raise ImportError( - "High-dimensional decision boundary projection requires the deeplearning extra: " - "install with 'pip install stride-xai[deeplearning]' (requires tensorflow)." + + ssnp = SSNP(epochs=ssnp_epochs, patience=ssnp_patience, verbose=0) + ssnp.fit(X_before_scaled, self.y_before) + except Exception as err: + raise OptionalDependencyError( + package_name="tensorflow", + feature_name="High-dimensional decision boundary projection (SSNP)", + extra_name="deeplearning", ) from err - # SSNP is used to find a 2D projection that preserves class structure. - ssnp = SSNP(epochs=ssnp_epochs, patience=ssnp_patience, verbose=0) - ssnp.fit(X_before_scaled, self.y_before) # Project points to 2D (if 2D already, this just returns the scaled data) X_before_2d = ssnp.transform(X_before_scaled) diff --git a/src/stride/xai/boundary/ssnp.py b/src/stride/xai/boundary/ssnp.py index 044c593..0265094 100644 --- a/src/stride/xai/boundary/ssnp.py +++ b/src/stride/xai/boundary/ssnp.py @@ -3,12 +3,21 @@ import os import numpy as np from sklearn.preprocessing import LabelBinarizer -import tensorflow as tf -from tensorflow.keras import regularizers -from tensorflow.keras.callbacks import EarlyStopping -from tensorflow.keras.initializers import Constant -from tensorflow.keras.layers import Dense, Input -from tensorflow.keras.models import Model + +from stride.exceptions import OptionalDependencyError + +try: + import tensorflow as tf + from tensorflow.keras import regularizers + from tensorflow.keras.callbacks import EarlyStopping + from tensorflow.keras.initializers import Constant + from tensorflow.keras.layers import Dense, Input + from tensorflow.keras.models import Model + + _HAS_TF = True +except ImportError: + _HAS_TF = False + tf = None # Ensure deterministic operations where possible os.environ["TF_DETERMINISTIC_OPS"] = "1" @@ -54,6 +63,13 @@ def __init__( self.inv = None self.clustering = None + if not _HAS_TF: + raise OptionalDependencyError( + package_name="tensorflow", + feature_name="SSNP boundary projection", + extra_name="deeplearning", + ) + tf.random.set_seed(42) tf.keras.backend.clear_session() diff --git a/src/stride/xai/clustering/__init__.py b/src/stride/xai/clustering/__init__.py index bf82546..9b0da06 100644 --- a/src/stride/xai/clustering/__init__.py +++ b/src/stride/xai/clustering/__init__.py @@ -1,7 +1,11 @@ from .clustering import ClusterBasedDriftDetector # noqa: F401 -from .visualization import ( - plot_drift_clustered, - plot_clusters_by_class, - plot_centers_shift, - plot_clustering_heatmap, -) # noqa: F401 + +try: + from .visualization import ( + plot_drift_clustered, + plot_clusters_by_class, + plot_centers_shift, + plot_clustering_heatmap, + ) # noqa: F401 +except ImportError: + pass diff --git a/src/stride/xai/clustering/xmeans.py b/src/stride/xai/clustering/xmeans.py index 51a1859..4feceff 100644 --- a/src/stride/xai/clustering/xmeans.py +++ b/src/stride/xai/clustering/xmeans.py @@ -4,7 +4,7 @@ from typing import Sequence import numpy as np -from pyclustering.cluster.xmeans import kmeans_plusplus_initializer, xmeans # type: ignore +from stride.exceptions import OptionalDependencyError def reshape_clusters(clusters: Sequence[Sequence[int]]) -> np.ndarray: @@ -63,6 +63,15 @@ def run_xmeans( here seeds both ``numpy.random`` and the built-in ``random`` module, which is sufficient for pyclustering's internal sampling. """ + try: + from pyclustering.cluster.xmeans import kmeans_plusplus_initializer, xmeans + except ImportError as err: + raise OptionalDependencyError( + package_name="pyclustering", + feature_name="X-Means clustering", + extra_name="clustering", + ) from err + if random_state is not None: random.seed(random_state) np.random.seed(random_state) diff --git a/src/stride/xai/importance/__init__.py b/src/stride/xai/importance/__init__.py index 0b6ff1e..c2d86ca 100644 --- a/src/stride/xai/importance/__init__.py +++ b/src/stride/xai/importance/__init__.py @@ -1,7 +1,11 @@ from .base import FeatureImportanceMethod # noqa: F401 from .methods import calculate_feature_importance # noqa: F401 -from .visualization import ( # noqa: F401 - visualize_drift_importance, - visualize_predictive_importance_shift, -) + +try: + from .visualization import ( # noqa: F401 + visualize_drift_importance, + visualize_predictive_importance_shift, + ) +except ImportError: + pass from .analysis import FeatureImportanceDriftAnalyzer # noqa: F401 diff --git a/src/stride/xai/importance/methods.py b/src/stride/xai/importance/methods.py index 494079f..680b5e9 100644 --- a/src/stride/xai/importance/methods.py +++ b/src/stride/xai/importance/methods.py @@ -1,7 +1,6 @@ import numpy as np from sklearn.inspection import permutation_importance -import shap -from lime.lime_tabular import LimeTabularExplainer +from stride.exceptions import OptionalDependencyError from .base import FeatureImportanceMethod @@ -68,6 +67,15 @@ def _calculate_pfi(model, X, y, n_repeats=30, random_state=42): def _calculate_shap(model, X, feature_names): """Calculate SHAP values.""" + try: + import shap + except ImportError as err: + raise OptionalDependencyError( + package_name="shap", + feature_name="SHAP feature importance", + extra_name="xai", + ) from err + # Use a subset for efficiency if dataset is large background_size = min(100, len(X)) background = shap.sample(X, background_size) @@ -123,6 +131,15 @@ def _calculate_shap(model, X, feature_names): def _calculate_lime(model, X, y, feature_names, random_state=42): """Calculate LIME feature importance.""" + try: + from lime.lime_tabular import LimeTabularExplainer + except ImportError as err: + raise OptionalDependencyError( + package_name="lime", + feature_name="LIME feature importance", + extra_name="xai", + ) from err + np.random.seed(random_state) # Create LIME explainer diff --git a/src/stride/xai/recurrence/methods.py b/src/stride/xai/recurrence/methods.py index 180ce6c..88c1d43 100644 --- a/src/stride/xai/recurrence/methods.py +++ b/src/stride/xai/recurrence/methods.py @@ -1,16 +1,11 @@ -import hdbscan import pandas as pd import numpy as np from sklearn.metrics import confusion_matrix from scipy.optimize import linear_sum_assignment +from stride.exceptions import OptionalDependencyError from stride.xai.recurrence.full_window_storage import FullWindowStorage -from stride.xai.recurrence.visualization import ( - visualize_distance_matrix, - show_distance_median, - plot_threshold_analysis_results, -) def median_mask(arr, k=3): @@ -27,6 +22,15 @@ def median_mask(arr, k=3): def cluster_windows(matrix: pd.DataFrame, fix_outliers=True, median_mask_width=1): + try: + import hdbscan + except ImportError as err: + raise OptionalDependencyError( + package_name="hdbscan", + feature_name="Window clustering (HDBSCAN)", + extra_name="clustering", + ) from err + clusterer = hdbscan.HDBSCAN( metric="precomputed", min_cluster_size=3, diff --git a/src/stride/xai/recurrence/protree/metrics/compare.py b/src/stride/xai/recurrence/protree/metrics/compare.py index b381472..271e104 100644 --- a/src/stride/xai/recurrence/protree/metrics/compare.py +++ b/src/stride/xai/recurrence/protree/metrics/compare.py @@ -2,7 +2,6 @@ import numpy as np import pandas as pd -from icecream import ic from stride.xai.recurrence.protree import TDataBatch, TPrototypes, TTarget from stride.xai.recurrence.protree.explainers.tree_distance import IExplainer @@ -430,7 +429,6 @@ def _one_way_swap_delta( new_accuracy = _get_accuracy(temp_prototypes, x, y, explainer) accuracy_changes.append(np.abs(baseline_accuracy - new_accuracy)) except Exception as e: - ic(prototypes) raise e if accuracy_changes: return np.mean(accuracy_changes) diff --git a/src/stride/xai/stats/__init__.py b/src/stride/xai/stats/__init__.py index 89329d9..c613df7 100644 --- a/src/stride/xai/stats/__init__.py +++ b/src/stride/xai/stats/__init__.py @@ -1,3 +1,7 @@ from .statistical_tests import StatisticalTestsDriftDetector, StatisticalTestType from .descriptive_statistics import DescriptiveStatisticsDriftDetector, StatisticsType -from .visualization import PlotOptions, plot_boxplot, plot_histogram, plot_kde, plot_ecdf, plot_violin, plot_qq + +try: + from .visualization import PlotOptions, plot_boxplot, plot_histogram, plot_kde, plot_ecdf, plot_violin, plot_qq +except ImportError: + pass From e4c2fbcab1e20789a9a48042be7e188c497d8d77 Mon Sep 17 00:00:00 2001 From: MichalRedm <45372892+MichalRedm@users.noreply.github.com> Date: Fri, 2 Oct 2026 12:48:57 +0200 Subject: [PATCH 2/3] test(api): add minimal installation and optional dependency guardrail tests - Validate clean top-level import of stride without optional packages in sys.modules - Validate core estimators and analyzers execute on pure base dependencies - Validate OptionalDependencyError triggers with actionable install hints for missing optional features --- tests/test_minimal_install_import.py | 224 +++++++++++++++++++++++++++ 1 file changed, 224 insertions(+) create mode 100644 tests/test_minimal_install_import.py diff --git a/tests/test_minimal_install_import.py b/tests/test_minimal_install_import.py new file mode 100644 index 0000000..93a11bb --- /dev/null +++ b/tests/test_minimal_install_import.py @@ -0,0 +1,224 @@ +"""Tests for verifying base minimal installation and optional dependency decoupling.""" + +import unittest +from unittest.mock import patch +import numpy as np +import pandas as pd + +from stride.common import DataDimensionsReducer, ReducerType +from stride.datasets.sea_drift import SeaDriftDataset +from stride.drift import BinaryErrorDriftDescriptor +from stride.exceptions import OptionalDependencyError +from stride.models import MLPModel, RandomForestModel +from stride.plotting.stream import plot_feature_space, visualize_data_stream +from stride.xai.boundary.analysis import DecisionBoundaryDriftAnalyzer +from stride.xai.boundary.ssnp import SSNP +from stride.xai.clustering.xmeans import run_xmeans +from stride.xai.importance.analysis import FeatureImportanceDriftAnalyzer +from stride.xai.importance.methods import calculate_feature_importance +from stride.xai.recurrence.methods import cluster_windows +from stride.xai.stats import ( + DescriptiveStatisticsDriftDetector, + StatisticalTestType, + StatisticalTestsDriftDetector, + StatisticsType, +) + + +# Mapping of optional third-party modules to mock as absent +MOCKED_ABSENT_OPTIONAL_MODULES = { + "shap": None, + "lime": None, + "lime.lime_tabular": None, + "umap": None, + "pyclustering": None, + "pyclustering.cluster.xmeans": None, + "hdbscan": None, + "tensorflow": None, + "matplotlib": None, + "matplotlib.pyplot": None, +} + + +class TestMinimalInstallImport(unittest.TestCase): + """Verify that importing and using core functionality succeeds in minimal base environment.""" + + def test_top_level_stride_import_without_optional_dependencies(self): + """Verify that 'import stride' succeeds when all optional libraries are uninstalled.""" + with patch.dict("sys.modules", MOCKED_ABSENT_OPTIONAL_MODULES): + import stride + + self.assertIsNotNone(stride.__version__) + self.assertTrue(hasattr(stride, "BinaryErrorDriftDescriptor")) + self.assertTrue(hasattr(stride, "RandomForestModel")) + self.assertTrue(hasattr(stride, "MLPModel")) + self.assertTrue(hasattr(stride, "DescriptiveStatisticsDriftDetector")) + self.assertTrue(hasattr(stride, "DecisionBoundaryDriftAnalyzer")) + self.assertTrue(hasattr(stride, "FeatureImportanceDriftAnalyzer")) + + def test_core_algorithms_execute_without_optional_dependencies(self): + """Verify that core models, datasets, statistics, and descriptors execute cleanly.""" + np.random.seed(42) + X = np.random.rand(60, 4) + y = np.random.randint(0, 2, 60) + + # 1. Core models + rf = RandomForestModel(n_estimators=5, random_state=42) + rf.fit(X, y) + self.assertGreaterEqual(rf.score(X, y), 0.0) + + mlp = MLPModel(max_iter=20, random_state=42) + mlp.fit(X, y) + self.assertGreaterEqual(mlp.score(X, y), 0.0) + + # 2. Core dataset generation (powered by promoted river engine) + sea = SeaDriftDataset() + X_df, y_s = sea.generate(n_samples_before=50, n_samples_after=50, drift_width=20) + self.assertEqual(len(X_df), 100) + self.assertEqual(len(y_s), 100) + + # 3. Core drift descriptor + descriptor = BinaryErrorDriftDescriptor() + self.assertIsNotNone(descriptor) + + # 4. Core statistical tests + df_before = pd.DataFrame(X[:30], columns=[f"f{i}" for i in range(4)]) + df_after = pd.DataFrame(X[30:], columns=[f"f{i}" for i in range(4)]) + y_before = y[:30] + y_after = y[30:] + + stats_detector = DescriptiveStatisticsDriftDetector(df_before, y_before, df_after, y_after) + drift_flag, details = stats_detector.detect(StatisticsType.Mean) + self.assertIsInstance(drift_flag, (bool, np.bool_)) + + test_detector = StatisticalTestsDriftDetector(df_before, y_before, df_after, y_after) + drift_test_flag = test_detector.detect(StatisticalTestType.WassersteinDistance) + self.assertIsInstance(drift_test_flag, (bool, np.bool_)) + + # 5. Core dimensionality reduction (PCA) + pca_reducer = DataDimensionsReducer(ReducerType.PCA, n_components=2) + X_reduced = pca_reducer.fit_transform(X) + self.assertEqual(X_reduced.shape, (60, 2)) + + # 6. Core feature importance (Permutation / PFI) + pfi_res = calculate_feature_importance(rf, X, y, method="permutation", n_repeats=2, random_state=42) + self.assertEqual(pfi_res["method"], "PFI") + self.assertEqual(len(pfi_res["importances_mean"]), 4) + + # 7. Core 2D decision boundary drift analyzer (uses DummyProjector, no TF required) + analyzer_2d = DecisionBoundaryDriftAnalyzer(X[:30, :2], y[:30], X[30:, :2], y[30:], random_state=42) + res_boundary = analyzer_2d.analyze() + self.assertIn("pre", res_boundary) + self.assertIn("post", res_boundary) + self.assertTrue(res_boundary["is_2d"]) + + +class TestOptionalDependencyGuardrails(unittest.TestCase): + """Verify that attempting to invoke optional features raises actionable OptionalDependencyError.""" + + def setUp(self): + np.random.seed(42) + self.X = np.random.rand(40, 4) + self.y = np.random.randint(0, 2, 40) + self.rf = RandomForestModel(n_estimators=5, random_state=42) + self.rf.fit(self.X, self.y) + + def test_shap_missing_raises_optional_dependency_error(self): + with patch.dict("sys.modules", {"shap": None}): + with self.assertRaises(OptionalDependencyError) as ctx: + calculate_feature_importance(self.rf, self.X, self.y, method="shap") + self.assertEqual(ctx.exception.package_name, "shap") + self.assertEqual(ctx.exception.extra_name, "xai") + self.assertIn("pip install stride-xai[xai]", str(ctx.exception)) + + def test_lime_missing_raises_optional_dependency_error(self): + with patch.dict("sys.modules", {"lime": None, "lime.lime_tabular": None}): + with self.assertRaises(OptionalDependencyError) as ctx: + calculate_feature_importance(self.rf, self.X, self.y, method="lime") + self.assertEqual(ctx.exception.package_name, "lime") + self.assertEqual(ctx.exception.extra_name, "xai") + self.assertIn("pip install stride-xai[xai]", str(ctx.exception)) + + def test_feature_importance_analyzer_shap_raises_optional_dependency_error(self): + with patch.dict("sys.modules", {"shap": None}): + analyzer = FeatureImportanceDriftAnalyzer(self.X[:20], self.y[:20], self.X[20:], self.y[20:]) + with self.assertRaises(OptionalDependencyError) as ctx: + analyzer.compute_drift_importance(importance_method="shap") + self.assertEqual(ctx.exception.package_name, "shap") + self.assertIn("pip install stride-xai[xai]", str(ctx.exception)) + + def test_umap_missing_raises_optional_dependency_error(self): + with patch.dict("sys.modules", {"umap": None}): + with self.assertRaises(OptionalDependencyError) as ctx: + DataDimensionsReducer(ReducerType.UMAP, n_components=2) + self.assertEqual(ctx.exception.package_name, "umap-learn") + self.assertEqual(ctx.exception.extra_name, "clustering") + self.assertIn("pip install stride-xai[clustering]", str(ctx.exception)) + + def test_xmeans_missing_raises_optional_dependency_error(self): + with patch.dict("sys.modules", {"pyclustering": None, "pyclustering.cluster.xmeans": None}): + with self.assertRaises(OptionalDependencyError) as ctx: + run_xmeans(self.X, k_init=2, k_max=4) + self.assertEqual(ctx.exception.package_name, "pyclustering") + self.assertEqual(ctx.exception.extra_name, "clustering") + self.assertIn("pip install stride-xai[clustering]", str(ctx.exception)) + + def test_hdbscan_missing_raises_optional_dependency_error(self): + with patch.dict("sys.modules", {"hdbscan": None}): + matrix = pd.DataFrame(np.zeros((5, 5))) + with self.assertRaises(OptionalDependencyError) as ctx: + cluster_windows(matrix) + self.assertEqual(ctx.exception.package_name, "hdbscan") + self.assertEqual(ctx.exception.extra_name, "clustering") + self.assertIn("pip install stride-xai[clustering]", str(ctx.exception)) + + def test_tensorflow_ssnp_missing_raises_optional_dependency_error(self): + with patch.dict("sys.modules", {"tensorflow": None}): + # 1. Direct SSNP class instantiation + with patch("stride.xai.boundary.ssnp._HAS_TF", False): + with self.assertRaises(OptionalDependencyError) as ctx: + SSNP() + self.assertEqual(ctx.exception.package_name, "tensorflow") + self.assertEqual(ctx.exception.extra_name, "deeplearning") + self.assertIn("pip install stride-xai[deeplearning]", str(ctx.exception)) + + # 2. DecisionBoundaryDriftAnalyzer on high-dimensional (>2D) data + analyzer_hd = DecisionBoundaryDriftAnalyzer(self.X[:20], self.y[:20], self.X[20:], self.y[20:], random_state=42) + with self.assertRaises(OptionalDependencyError) as ctx: + analyzer_hd.analyze() + self.assertEqual(ctx.exception.package_name, "tensorflow") + self.assertEqual(ctx.exception.extra_name, "deeplearning") + self.assertIn("pip install stride-xai[deeplearning]", str(ctx.exception)) + + def test_matplotlib_stream_plotting_missing_raises_optional_dependency_error(self): + with patch("stride.plotting.stream._HAS_MATPLOTLIB", False): + with self.assertRaises(OptionalDependencyError) as ctx: + plot_feature_space( + n_features=2, + feature_names=["f1", "f2"], + X_before=self.X[:20, :2], + X_after=self.X[20:, :2], + y_before=self.y[:20], + y_after=self.y[20:], + class_colors={0: "red", 1: "blue"}, + ) + self.assertEqual(ctx.exception.package_name, "matplotlib") + self.assertEqual(ctx.exception.extra_name, "vis") + self.assertIn("pip install stride-xai[vis]", str(ctx.exception)) + + with self.assertRaises(OptionalDependencyError) as ctx: + visualize_data_stream( + X=self.X, + y=self.y, + window_before_start=0, + window_after_start=20, + window_length=20, + feature_names=["f1", "f2", "f3", "f4"], + ) + self.assertEqual(ctx.exception.package_name, "matplotlib") + self.assertEqual(ctx.exception.extra_name, "vis") + self.assertIn("pip install stride-xai[vis]", str(ctx.exception)) + + +if __name__ == "__main__": + unittest.main() From a413e4d4513880fbc81058e7195823afe922b90c Mon Sep 17 00:00:00 2001 From: MichalRedm <45372892+MichalRedm@users.noreply.github.com> Date: Fri, 2 Oct 2026 12:49:05 +0200 Subject: [PATCH 3/3] docs(agents): update project context milestone for decoupled optional dependencies --- .agents/project_context.md | 2 ++ 1 file changed, 2 insertions(+) diff --git a/.agents/project_context.md b/.agents/project_context.md index 5bf6e14..fe68c66 100644 --- a/.agents/project_context.md +++ b/.agents/project_context.md @@ -22,6 +22,8 @@ Develop, benchmark, and visualize Explainable AI (xAI) techniques for characteri - [x] Modular optional dependency extras (`vis`, `drift`, `clustering`, `xai`, `deeplearning`, `dashboard`, `dev`, `all`). - [x] Single Responsibility Principle (SRP) cleanup, decomposition of monolithic God Classes (`ClusterBasedDriftDetector`), and headless plotting execution without `plt.show()`. - [x] Standardized GitHub agent instructions, Git & PR workflow skill (`.agents/skills/git-pr-workflow/`), issue templates, and PR template aligned with institutional engineering standards. +- [x] Decouple optional dependency imports (`umap-learn`, `shap`, `lime`, `pyclustering`, `hdbscan`, `tensorflow`, `matplotlib`, `seaborn`) and establish minimal base installation guardrails (`OptionalDependencyError`). +- [ ] PyPI Packaging, Metadata Standardization, and Automated OIDC Trusted Publishing Workflow (#10). - [ ] Expand automated test coverage for core xAI algorithms in `src/stride/`. - [ ] Implement additional statistical drift detectors and recurring concept benchmarks.