diff --git a/configs/BENCH-CONFIG-SPEC.md b/configs/BENCH-CONFIG-SPEC.md index 26bff24f..0c2a7b56 100644 --- a/configs/BENCH-CONFIG-SPEC.md +++ b/configs/BENCH-CONFIG-SPEC.md @@ -108,6 +108,7 @@ Configs have the three highest parameter keys: |:---------------|:--------------|:--------|:------------| | `algorithm`:`estimator` | None | | Name of measured estimator. | | `algorithm`:`estimator_params` | Empty `dict` | | Parameters for estimator constructor. | +| `algorithm`:`method_params`:`{method}` | None | | Parameters for a measured method of estimator. A method listed here is called with these parameters only, without data, which fits methods working off the fitted model such as `HDBSCAN.dbscan_clustering`. | | `algorithm`:`batch_size`:`{stage}` | None | Any positive integer | Enables online mode for `{stage}` methods of estimator (sequential calls for each batch). | | `algorithm`:`sklearn_context` | None | | Parameters for sklearn `config_context` used over estimator. `array_api_dispatch` requires `SCIPY_ARRAY_API=1`, which scikit-learn_bench sets by default if it is unset in the environment. | | `algorithm`:`sklearnex_context` | None | | Parameters for sklearnex `config_context` used over estimator. Updated by `sklearn_context` if set. | @@ -140,6 +141,7 @@ List of available special values: | `algorithm`:`estimator_params`:`scale_pos_weight` | sklearn_estimator | `auto` | Sets `scale_pos_weight` parameter to `sum(negative instances) / sum(positive instances)` value for estimator. | | `algorithm`:`estimator_params`:`n_clusters` | sklearn_estimator | `auto` | Sets `n_clusters` parameter to number of clusters or classes from dataset description for estimator. | | `algorithm`:`estimator_params`:`eps` | sklearn_estimator | `distances_quantile:{quantile}` format where quantile is *float* value in [0, 1] range | Computes `eps` parameter as quantile value of distances in `x_train` matrix for estimator. | +| `algorithm`:`method_params`:`dbscan_clustering`:`cut_distance` | sklearn_estimator | `distances_quantile:{quantile}` format where quantile is *float* value in [0, 1] range | Computes `cut_distance` the same way as `eps` above, so that an HDBSCAN hierarchy can be re-cut at the distance DBSCAN is measured at. | ## Range of Values diff --git a/configs/common/hdbscan.json b/configs/common/hdbscan.json new file mode 100644 index 00000000..fd82b70d --- /dev/null +++ b/configs/common/hdbscan.json @@ -0,0 +1,48 @@ +{ + "PARAMETERS_SETS": { + "common hdbscan parameters": { + "algorithm": { + "estimator": "HDBSCAN", + "estimator_params": { + "min_cluster_size": 5, + "min_samples": 5, + "metric": "euclidean", + "cluster_selection_method": "eom", + "allow_single_cluster": false, + "store_centers": null, + "copy": false + }, + "estimator_methods": { "training": "fit" } + }, + "data": { "format": "numpy", "order": "C", "dtype": "float64" }, + "bench": { "n_runs": 3, "time_limit": 1200 } + }, + "sklearn hdbscan parameters": { + "algorithm": { + "estimator_params": { "n_jobs": "[SPECIAL_VALUE]physical_cpus" } + } + }, + "hdbscan brute method": { + "algorithm": { "estimator_params": { "algorithm": "brute" } } + }, + "hdbscan kd_tree method": { + "algorithm": { "estimator_params": { "algorithm": "kd_tree", "leaf_size": 40 } } + }, + "hdbscan ball_tree method": { + "algorithm": { "estimator_params": { "algorithm": "ball_tree", "leaf_size": 40 } } + }, + "hdbscan tree methods": { + "algorithm": { + "estimator_params": { "algorithm": ["kd_tree", "ball_tree"], "leaf_size": 40 } + } + }, + "hdbscan all methods": { + "algorithm": { + "estimator_params": { + "algorithm": ["brute", "kd_tree", "ball_tree"], + "leaf_size": 40 + } + } + } + } +} diff --git a/configs/experiments/README.md b/configs/experiments/README.md index 2b6225c5..dc442c9b 100644 --- a/configs/experiments/README.md +++ b/configs/experiments/README.md @@ -2,4 +2,8 @@ `daal4py_svd`: tests performance scalability of `daal4py.svd` algorithm +`hdbscan_parameters`: sweeps the `HDBSCAN` parameter space (metrics, cluster selection, density thresholds, stored centers, dtypes and data formats) over the `sklearn` and `sklearnex` implementations. + +`hdbscan_scaling`: tests thread, NUMA, `n_samples` and `n_features` scalability of `HDBSCAN`. + `nearest_neighbors`: tests performance of neighbors search implementations from `sklearnex`, `sklearn`, `raft`, `faiss` and `svs`. diff --git a/configs/experiments/hdbscan_parameters.json b/configs/experiments/hdbscan_parameters.json new file mode 100644 index 00000000..8e1f57ee --- /dev/null +++ b/configs/experiments/hdbscan_parameters.json @@ -0,0 +1,169 @@ +{ + "INCLUDE": ["../common/sklearn.json", "../common/hdbscan.json"], + "PARAMETERS_SETS": { + "hdbscan tree metrics": { + "algorithm": { + "estimator_params": { "metric": ["euclidean", "manhattan", "chebyshev"] } + } + }, + "hdbscan minkowski metric": { + "algorithm": { + "estimator_params": { "metric": "minkowski", "metric_params": { "p": 3 } } + } + }, + "hdbscan cosine metric": { + "algorithm": { + "estimator_params": { "metric": "cosine", "algorithm": "brute" } + } + }, + "hdbscan cluster selection sweep": { + "algorithm": { + "estimator_params": { "cluster_selection_method": ["eom", "leaf"] } + } + }, + "hdbscan density sweep": { + "algorithm": { + "estimator_params": { "min_cluster_size": [5, 25, 100], "min_samples": [5, 25] } + } + }, + "hdbscan store centers sweep": { + "algorithm": { + "estimator_params": { "store_centers": [null, "centroid", "medoid", "both"] } + } + }, + "hdbscan dtype sweep": { + "data": { "dtype": ["float32", "float64"] } + }, + "hdbscan data format sweep": { + "data": [ + { "format": "numpy", "order": "C" }, + { "format": "numpy", "order": "F" }, + { "format": "pandas", "order": "F" } + ] + }, + "hdbscan parameters data [tree]": { + "data": { + "source": "make_blobs", + "generation_kwargs": { + "centers": 10, + "n_samples": 100000, + "n_features": 8, + "cluster_std": 2.0, + "random_state": 42 + }, + "split_kwargs": { "ignore": true } + } + }, + "hdbscan parameters data [brute]": { + "data": { + "source": "make_blobs", + "generation_kwargs": { + "centers": 10, + "n_samples": 25000, + "n_features": 64, + "cluster_std": 4.0, + "random_state": 42 + }, + "split_kwargs": { "ignore": true } + } + }, + "hdbscan parameters data [real]": { + "data": { + "dataset": "skin_segmentation", + "split_kwargs": { "train_size": 30000 }, + "preprocessing_kwargs": { "normalize": "standard" } + } + } + }, + "TEMPLATES": { + "hdbscan metrics [tree]": { + "SETS": [ + "sklearn-ex[cpu] implementations", + "common hdbscan parameters", + "sklearn hdbscan parameters", + "hdbscan tree methods", + "hdbscan tree metrics", + "hdbscan parameters data [tree]" + ] + }, + "hdbscan minkowski [tree]": { + "SETS": [ + "sklearn-ex[cpu] implementations", + "common hdbscan parameters", + "sklearn hdbscan parameters", + "hdbscan kd_tree method", + "hdbscan minkowski metric", + "hdbscan parameters data [tree]" + ] + }, + "hdbscan metrics [brute]": { + "SETS": [ + "sklearn-ex[cpu] implementations", + "common hdbscan parameters", + "sklearn hdbscan parameters", + "hdbscan brute method", + "hdbscan tree metrics", + "hdbscan parameters data [brute]" + ] + }, + "hdbscan cosine [brute]": { + "SETS": [ + "sklearn-ex[cpu] implementations", + "common hdbscan parameters", + "sklearn hdbscan parameters", + "hdbscan cosine metric", + "hdbscan parameters data [brute]" + ] + }, + "hdbscan cluster selection": { + "SETS": [ + "sklearn-ex[cpu] implementations", + "common hdbscan parameters", + "sklearn hdbscan parameters", + "hdbscan kd_tree method", + "hdbscan cluster selection sweep", + "hdbscan parameters data [tree]" + ] + }, + "hdbscan density": { + "SETS": [ + "sklearn-ex[cpu] implementations", + "common hdbscan parameters", + "sklearn hdbscan parameters", + "hdbscan kd_tree method", + "hdbscan density sweep", + "hdbscan parameters data [tree]" + ] + }, + "hdbscan store centers": { + "SETS": [ + "sklearn-ex[cpu] implementations", + "common hdbscan parameters", + "sklearn hdbscan parameters", + "hdbscan kd_tree method", + "hdbscan store centers sweep", + "hdbscan parameters data [tree]" + ] + }, + "hdbscan dtypes": { + "SETS": [ + "sklearn-ex[cpu] implementations", + "common hdbscan parameters", + "sklearn hdbscan parameters", + "hdbscan all methods", + "hdbscan dtype sweep", + "hdbscan parameters data [real]" + ] + }, + "hdbscan data formats": { + "SETS": [ + "sklearn-ex[cpu] implementations", + "common hdbscan parameters", + "sklearn hdbscan parameters", + "hdbscan kd_tree method", + "hdbscan data format sweep", + "hdbscan parameters data [tree]" + ] + } + } +} diff --git a/configs/experiments/hdbscan_scaling.json b/configs/experiments/hdbscan_scaling.json new file mode 100644 index 00000000..997609a6 --- /dev/null +++ b/configs/experiments/hdbscan_scaling.json @@ -0,0 +1,217 @@ +{ + "INCLUDE": ["../common/sklearn.json", "../common/hdbscan.json"], + "PARAMETERS_SETS": { + "sklearnex hdbscan implementation": { + "algorithm": { "library": "sklearnex", "device": "cpu" } + }, + "hdbscan thread scaling": [ + { + "bench": { "taskset": "0" }, + "algorithm": { "estimator_params": { "n_jobs": 1 } } + }, + { + "bench": { "taskset": "0-3" }, + "algorithm": { "estimator_params": { "n_jobs": 4 } } + }, + { + "bench": { "taskset": "0-7" }, + "algorithm": { "estimator_params": { "n_jobs": 8 } } + }, + { + "bench": { "taskset": "0-15" }, + "algorithm": { "estimator_params": { "n_jobs": 16 } } + }, + { + "bench": { "taskset": "0-27" }, + "algorithm": { "estimator_params": { "n_jobs": 28 } } + }, + { + "bench": { "taskset": "0-55" }, + "algorithm": { "estimator_params": { "n_jobs": 56 } } + }, + { + "bench": { "taskset": "0-111" }, + "algorithm": { "estimator_params": { "n_jobs": 112 } } + }, + { + "bench": { "taskset": "0-223" }, + "algorithm": { "estimator_params": { "n_jobs": 224 } } + } + ], + "hdbscan numa scaling": [ + { + "bench": { "taskset": "0-55" }, + "algorithm": { "estimator_params": { "n_jobs": 56 } } + }, + { + "bench": { "taskset": "0-27,56-83" }, + "algorithm": { "estimator_params": { "n_jobs": 56 } } + }, + { + "bench": { "taskset": "0-111" }, + "algorithm": { "estimator_params": { "n_jobs": 112 } } + } + ], + "hdbscan scaling profiling": { + "bench": { + "n_runs": 3, + "time_limit": 1200, + "cpu_profile": true, + "memory_profile": true + } + }, + "hdbscan brute scaling data": { + "data": { + "source": "make_blobs", + "generation_kwargs": { + "centers": 10, + "n_samples": 50000, + "n_features": 256, + "cluster_std": 4.0, + "random_state": 42 + }, + "split_kwargs": { "ignore": true } + } + }, + "hdbscan tree scaling data": { + "data": { + "source": "make_blobs", + "generation_kwargs": { + "centers": 10, + "n_samples": 1000000, + "n_features": 16, + "cluster_std": 2.0, + "random_state": 42 + }, + "split_kwargs": { "ignore": true } + } + }, + "hdbscan n_samples sweep [brute]": { + "data": { + "source": "make_blobs", + "generation_kwargs": { + "centers": 10, + "n_samples": "[RANGE]mul:5000:40000:2", + "n_features": 64, + "cluster_std": 4.0, + "random_state": 42 + }, + "split_kwargs": { "ignore": true } + } + }, + "hdbscan n_samples sweep [tree]": { + "data": { + "source": "make_blobs", + "generation_kwargs": { + "centers": 10, + "n_samples": "[RANGE]mul:25000:400000:2", + "n_features": 8, + "cluster_std": 2.0, + "random_state": 42 + }, + "split_kwargs": { "ignore": true } + } + }, + "hdbscan n_samples sweep [tree, sklearnex only]": { + "data": { + "source": "make_blobs", + "generation_kwargs": { + "centers": 10, + "n_samples": "[RANGE]mul:500000:8000000:2", + "n_features": 8, + "cluster_std": 2.0, + "random_state": 42 + }, + "split_kwargs": { "ignore": true } + } + }, + "hdbscan n_features sweep": { + "data": { + "source": "make_blobs", + "generation_kwargs": { + "centers": 10, + "n_samples": 20000, + "n_features": "[RANGE]pow:2:2:10", + "cluster_std": 4.0, + "random_state": 42 + }, + "split_kwargs": { "ignore": true } + } + }, + "hdbscan size sweep limits": { + "bench": { "n_runs": 3, "time_limit": 300 } + } + }, + "TEMPLATES": { + "hdbscan thread scaling [brute]": { + "SETS": [ + "sklearnex hdbscan implementation", + "common hdbscan parameters", + "hdbscan scaling profiling", + "hdbscan brute method", + "hdbscan thread scaling", + "hdbscan brute scaling data" + ] + }, + "hdbscan thread scaling [kd_tree]": { + "SETS": [ + "sklearnex hdbscan implementation", + "common hdbscan parameters", + "hdbscan scaling profiling", + "hdbscan kd_tree method", + "hdbscan thread scaling", + "hdbscan tree scaling data" + ] + }, + "hdbscan numa scaling [brute]": { + "SETS": [ + "sklearnex hdbscan implementation", + "common hdbscan parameters", + "hdbscan scaling profiling", + "hdbscan brute method", + "hdbscan numa scaling", + "hdbscan brute scaling data" + ] + }, + "hdbscan n_samples scaling [brute]": { + "SETS": [ + "sklearn-ex[cpu] implementations", + "common hdbscan parameters", + "sklearn hdbscan parameters", + "hdbscan brute method", + "hdbscan n_samples sweep [brute]", + "hdbscan size sweep limits" + ] + }, + "hdbscan n_samples scaling [kd_tree]": { + "SETS": [ + "sklearn-ex[cpu] implementations", + "common hdbscan parameters", + "sklearn hdbscan parameters", + "hdbscan kd_tree method", + "hdbscan n_samples sweep [tree]", + "hdbscan size sweep limits" + ] + }, + "hdbscan n_samples scaling [kd_tree, beyond sklearn]": { + "SETS": [ + "sklearnex hdbscan implementation", + "common hdbscan parameters", + "sklearn hdbscan parameters", + "hdbscan kd_tree method", + "hdbscan n_samples sweep [tree, sklearnex only]", + "hdbscan size sweep limits" + ] + }, + "hdbscan n_features scaling": { + "SETS": [ + "sklearn-ex[cpu] implementations", + "common hdbscan parameters", + "sklearn hdbscan parameters", + "hdbscan all methods", + "hdbscan n_features sweep", + "hdbscan size sweep limits" + ] + } + } +} diff --git a/configs/experiments/hdbscan_vs_dbscan.json b/configs/experiments/hdbscan_vs_dbscan.json new file mode 100644 index 00000000..a8a876d3 --- /dev/null +++ b/configs/experiments/hdbscan_vs_dbscan.json @@ -0,0 +1,123 @@ +{ + "INCLUDE": ["../common/sklearn.json", "../common/hdbscan.json", "../common/dbscan.json"], + "PARAMETERS_SETS": { + "hdbscan re-cut as dbscan": { + "algorithm": { + "estimator_methods": { "training": "fit|dbscan_clustering" }, + "method_params": { + "dbscan_clustering": { + "cut_distance": "[SPECIAL_VALUE]distances_quantile:0.01", + "min_cluster_size": 5 + } + } + }, + "bench": { "ensure_sklearnex_patching": false } + }, + "density comparison datasets [tree]": { + "data": [ + { + "dataset": "skin_segmentation", + "split_kwargs": { "train_size": 100000 }, + "preprocessing_kwargs": { "normalize": "standard" } + }, + { + "source": "make_blobs", + "generation_kwargs": { + "centers": 10, + "n_samples": 50000, + "n_features": 8, + "cluster_std": 2.0, + "random_state": 42 + }, + "split_kwargs": { "ignore": true } + } + ] + }, + "density comparison datasets [brute]": { + "data": [ + { + "dataset": "mnist", + "split_kwargs": { "train_size": 20000 }, + "preprocessing_kwargs": { "normalize": "standard" } + }, + { + "source": "make_blobs", + "generation_kwargs": { + "centers": 10, + "n_samples": 25000, + "n_features": 64, + "cluster_std": 4.0, + "random_state": 42 + }, + "split_kwargs": { "ignore": true } + } + ] + }, + "density comparison data format": { + "data": { "format": "numpy", "order": "C", "dtype": "float64" }, + "bench": { "n_runs": 3, "time_limit": 1200 } + } + }, + "TEMPLATES": { + "dbscan [tree-friendly data]": { + "SETS": [ + "common dbscan parameters", + "sklearn dbscan parameters", + "density comparison datasets [tree]", + "density comparison data format", + "sklearn-ex[cpu,gpu] implementations" + ] + }, + "dbscan [brute-friendly data]": { + "SETS": [ + "common dbscan parameters", + "sklearn dbscan parameters", + "density comparison datasets [brute]", + "density comparison data format", + "sklearn-ex[cpu,gpu] implementations" + ] + }, + "hdbscan [tree-friendly data]": { + "SETS": [ + "common hdbscan parameters", + "sklearn hdbscan parameters", + "hdbscan kd_tree method", + "density comparison datasets [tree]", + "density comparison data format", + "sklearn-ex[cpu,gpu] implementations" + ] + }, + "hdbscan [brute-friendly data]": { + "SETS": [ + "common hdbscan parameters", + "sklearn hdbscan parameters", + "hdbscan brute method", + "density comparison datasets [brute]", + "density comparison data format", + "sklearn-ex[cpu,gpu] implementations" + ] + }, + "hdbscan re-cut at the dbscan distance [tree-friendly data]": { + "SETS": [ + "common hdbscan parameters", + "sklearn hdbscan parameters", + "hdbscan kd_tree method", + "hdbscan re-cut as dbscan", + "density comparison datasets [tree]", + "density comparison data format", + "sklearn-ex[cpu] implementations" + ] + }, + "hdbscan re-cut at the dbscan distance [brute-friendly data]": { + "SETS": [ + "common hdbscan parameters", + "sklearn hdbscan parameters", + "hdbscan brute method", + "hdbscan re-cut as dbscan", + "density comparison datasets [brute]", + "density comparison data format", + "sklearn-ex[cpu] implementations" + ] + } + } +} diff --git a/configs/regular/hdbscan.json b/configs/regular/hdbscan.json new file mode 100644 index 00000000..1f2bb69f --- /dev/null +++ b/configs/regular/hdbscan.json @@ -0,0 +1,63 @@ +{ + "INCLUDE": ["../common/sklearn.json", "../common/hdbscan.json"], + "PARAMETERS_SETS": { + "hdbscan tree-friendly datasets": { + "data": [ + { + "dataset": "skin_segmentation", + "split_kwargs": { "train_size": 100000 }, + "preprocessing_kwargs": { "normalize": "standard" } + }, + { + "source": "make_blobs", + "generation_kwargs": { + "centers": 100, + "n_samples": 100000, + "n_features": 8, + "cluster_std": 1.0 + }, + "split_kwargs": { "ignore": true } + } + ] + }, + "hdbscan brute-friendly datasets": { + "data": [ + { + "dataset": "mnist", + "split_kwargs": { "train_size": 20000 }, + "preprocessing_kwargs": { "normalize": "standard" } + }, + { + "source": "make_blobs", + "generation_kwargs": { + "centers": 10, + "n_samples": 25000, + "n_features": 4, + "cluster_std": 1.0 + }, + "split_kwargs": { "ignore": true } + } + ] + } + }, + "TEMPLATES": { + "hdbscan kd_tree": { + "SETS": [ + "common hdbscan parameters", + "sklearn hdbscan parameters", + "hdbscan kd_tree method", + "hdbscan tree-friendly datasets", + "sklearn-ex[cpu,gpu] implementations" + ] + }, + "hdbscan brute force": { + "SETS": [ + "common hdbscan parameters", + "sklearn hdbscan parameters", + "hdbscan brute method", + "hdbscan brute-friendly datasets", + "sklearn-ex[cpu,gpu] implementations" + ] + } + } +} diff --git a/configs/weekly/hdbscan.json b/configs/weekly/hdbscan.json new file mode 100644 index 00000000..0c9dac0f --- /dev/null +++ b/configs/weekly/hdbscan.json @@ -0,0 +1,68 @@ +{ + "INCLUDE": ["../common/sklearn.json", "../common/hdbscan.json"], + "PARAMETERS_SETS": { + "high-load hdbscan tree-friendly datasets": { + "data": [ + { + "dataset": ["road_network", "covtype"], + "split_kwargs": { "train_size": 100000 }, + "preprocessing_kwargs": { "normalize": "standard" } + }, + { + "source": "make_blobs", + "generation_kwargs": { + "centers": 10, + "n_samples": 100000, + "n_features": [4, 16, 64], + "cluster_std": 2.0 + }, + "split_kwargs": { "ignore": true } + } + ] + }, + "high-load hdbscan brute-friendly datasets": { + "data": [ + { + "dataset": "cifar", + "split_kwargs": { "train_size": 15000 }, + "preprocessing_kwargs": { "normalize": "mean" } + }, + { + "dataset": "sensit", + "split_kwargs": { "train_size": 25000 }, + "preprocessing_kwargs": { "normalize": "standard" } + }, + { + "source": "make_blobs", + "generation_kwargs": { + "centers": 10, + "n_samples": 25000, + "n_features": [64, 256, 1024], + "cluster_std": 4.0 + }, + "split_kwargs": { "ignore": true } + } + ] + } + }, + "TEMPLATES": { + "hdbscan tree methods": { + "SETS": [ + "sklearn-ex[cpu] implementations", + "common hdbscan parameters", + "sklearn hdbscan parameters", + "hdbscan tree methods", + "high-load hdbscan tree-friendly datasets" + ] + }, + "hdbscan brute force": { + "SETS": [ + "sklearn-ex[cpu] implementations", + "common hdbscan parameters", + "sklearn hdbscan parameters", + "hdbscan brute method", + "high-load hdbscan brute-friendly datasets" + ] + } + } +} diff --git a/sklbench/benchmarks/common.py b/sklbench/benchmarks/common.py index 1df1e1a5..bd7b3376 100644 --- a/sklbench/benchmarks/common.py +++ b/sklbench/benchmarks/common.py @@ -16,6 +16,7 @@ import argparse import json +import re from typing import Dict from ..utils.bench_case import get_bench_case_value, get_data_name @@ -26,16 +27,23 @@ def enrich_result(result: Dict, bench_case: BenchCase) -> Dict: """Common function for all benchmarks to update the result with additional information""" + library = ( + get_bench_case_value(bench_case, "algorithm:library") + .replace( + # skipping emulators namespace for conciseness + "sklbench.emulators.", + "", + ) + .replace(".utils", "") + ) + # estimators available only from a preview namespace (like `sklearnex.preview.cluster`) + # are reported under their library so that they are comparable + # with the stock implementation in the report + library = re.sub(r"\.preview(\..+)?$", "", library) result.update( { "dataset": get_data_name(bench_case, shortened=True), - "library": get_bench_case_value(bench_case, "algorithm:library") - .replace( - # skipping emulators namespace for conciseness - "sklbench.emulators.", - "", - ) - .replace(".utils", ""), + "library": library, "device": get_bench_case_value(bench_case, "algorithm:device"), } ) diff --git a/sklbench/benchmarks/sklearn_estimator.py b/sklbench/benchmarks/sklearn_estimator.py index 8c511ce3..505e2f02 100644 --- a/sklbench/benchmarks/sklearn_estimator.py +++ b/sklbench/benchmarks/sklearn_estimator.py @@ -114,6 +114,31 @@ def get_number_of_classes(estimator_instance, y): return len(np.unique(y)) +def get_clustering_metrics_of_labels(labels, x_compat, y_compat) -> Dict[str, float]: + """Score a cluster assignment produced by a clustering estimator. + + @param labels Cluster index per sample, `-1` for noise + @param x_compat Clustered data as a numpy array + @param y_compat Ground truth labels as a numpy array + + @return Cluster count and, for more than one cluster, the Davies-Bouldin, + homogeneity and completeness scores + """ + labels = convert_to_numpy(labels) + clusters = len(np.unique(labels[labels != -1])) + metrics = {"clusters": clusters} + if clusters > 1: + metrics["Davies-Bouldin score"] = float(davies_bouldin_score(x_compat, labels)) + if len(np.unique(y_compat)) < 128: + metrics["homogeneity"] = ( + float(homogeneity_score(y_compat, labels)) if clusters > 1 else 0 + ) + metrics["completeness"] = ( + float(completeness_score(y_compat, labels)) if clusters > 1 else 0 + ) + return metrics + + def get_subset_metrics_of_estimator( task, stage, estimator_instance, data ) -> Dict[str, float]: @@ -214,32 +239,11 @@ def get_subset_metrics_of_estimator( } ) if "DBSCAN" in str(estimator_instance) and stage == "training": - labels = convert_to_numpy(estimator_instance.labels_) - clusters = len(np.unique(labels[labels != -1])) - metrics.update({"clusters": clusters}) - if clusters > 1: - metrics.update( - { - "Davies-Bouldin score": float( - davies_bouldin_score(x_compat, labels) - ) - } - ) - if len(np.unique(y)) < 128: - metrics.update( - { - "homogeneity": ( - float(homogeneity_score(y_compat, labels)) - if clusters > 1 - else 0 - ), - "completeness": ( - float(completeness_score(y_compat, labels)) - if clusters > 1 - else 0 - ), - } + metrics.update( + get_clustering_metrics_of_labels( + estimator_instance.labels_, x_compat, y_compat ) + ) elif task == "manifold": if hasattr(estimator_instance, "kl_divergence_") and stage == "training": metrics.update( @@ -444,12 +448,23 @@ def measure_sklearn_estimator( sklearnex_logging_stream = get_sklearnex_logging_stream() metrics = dict() + # quality metrics of the labels a data-free method returns itself, which the + # per-stage metrics below cannot see + method_label_metrics = dict() estimator_instance = estimator_class(**estimator_params) for stage in estimator_methods.keys(): for method in estimator_methods[stage]: if hasattr(estimator_instance, method): method_instance = getattr(estimator_instance, method) - if "y" in list(inspect.signature(method_instance).parameters): + # a method given its own arguments takes no data: it works off the + # fitted model instead, like 'HDBSCAN.dbscan_clustering', which + # re-cuts the hierarchy 'fit' built at another distance + method_params = get_bench_case_value( + bench_case, f"algorithm:method_params:{method}", None + ) + if method_params is not None: + data_args = () + elif "y" in list(inspect.signature(method_instance).parameters): if stage == "training": data_args = (x_train, y_train) else: @@ -484,7 +499,17 @@ def measure_sklearn_estimator( ) method_instance = getattr(daal_model, method) - metrics[method] = measure_case(bench_case, method_instance, *data_args) + metrics[method] = measure_case( + bench_case, method_instance, *data_args, **(method_params or {}) + ) + if method_params is not None and task == "clustering": + # `labels_` still holds the 'fit' partition, so the returned + # labels are the only description of what this method did + method_label_metrics[method] = get_clustering_metrics_of_labels( + getattr(estimator_instance, method)(**method_params), + convert_to_numpy(x_train), + convert_to_numpy(y_train), + ) if ensure_sklearnex_patching: full_method_name = f"{estimator_class.__name__}.{method}" sklearnex_logging_stream.seek(0) @@ -508,6 +533,8 @@ def measure_sklearn_estimator( for stage in estimator_methods.keys(): if method in estimator_methods[stage]: metrics[method].update(quality_metrics[stage]) + if method in method_label_metrics: + metrics[method].update(method_label_metrics[method]) return metrics, estimator_instance diff --git a/sklbench/utils/measurement.py b/sklbench/utils/measurement.py index 4df6c57b..66e1f100 100644 --- a/sklbench/utils/measurement.py +++ b/sklbench/utils/measurement.py @@ -56,6 +56,11 @@ def box_filter(array, left=0.2, right=0.8): return array[0], 0.0 lower, upper = array[int(size * left)], array[int(size * right)] result = np.array([item for item in array if lower < item < upper]) + if result.size == 0: + # the box is empty for short series (2 measurements always are, and a + # `time_limit` early stop can leave exactly 2), which would make the + # aggregated time NaN - fall back to the whole series instead + result = np.array(array) return np.mean(result), np.std(result) diff --git a/sklbench/utils/special_params.py b/sklbench/utils/special_params.py index 80812012..95b0bceb 100644 --- a/sklbench/utils/special_params.py +++ b/sklbench/utils/special_params.py @@ -148,6 +148,32 @@ def get_ratio_from_n_jobs(n_jobs: str) -> float: raise ValueError(f'Wrong arguments {args} in "n_jobs" special value') +def get_distances_quantile(x_train, special_value: str) -> float: + """Computes a quantile of the pairwise euclidean distances of `x_train`. + + Args: + x_train: Training data in any of the supported formats. + special_value: Special value of the form `distances_quantile:{quantile}`. + + Returns: + The requested quantile of the non-zero pairwise distances. + """ + x_train = convert_to_numpy(x_train) + quantile = float(special_value.replace(SP_VALUE_STR, "").split(":")[1]) + # subsample of x_train is used to avoid reaching of memory limit for large matrices + subsample = list(getattr(x_train, "index", np.arange(x_train.shape[0]))) + np.random.seed(42) + np.random.shuffle(subsample) + subsample = subsample[: min(x_train.shape[0], 1000)] + x_sample = x_train.loc[subsample] if hasattr(x_train, "loc") else x_train[subsample] + # conversion to lower precision is required + # to produce same distances quantile for different dtypes of x + x_sample = x_sample.astype("float32") + dist = np.tril(euclidean_distances(x_sample, x_sample)).reshape(-1) + dist = dist[dist != 0] + return float(np.quantile(dist, quantile)) + + def assign_case_special_values_on_run( bench_case: BenchCase, data, data_description: Dict ): @@ -270,25 +296,17 @@ def assign_case_special_values_on_run( "Unable to auto-assign n_clusters: " "data description doesn't have n_clusters or n_classes" ) - # "eps" auto assignment for DBSCAN - eps = get_bench_case_value(bench_case, "algorithm:estimator_params:eps", None) - if is_special_value(eps) and eps.replace(SP_VALUE_STR, "").startswith( - "distances_quantile" + # "eps" auto assignment for DBSCAN, and the equivalent "cut_distance" for the + # re-cut of an HDBSCAN hierarchy, so that both are the same distance on the + # same data and the two clusterings can be compared + for param_name in ( + "algorithm:estimator_params:eps", + "algorithm:method_params:dbscan_clustering:cut_distance", ): - x_train = convert_to_numpy(data[0]) - quantile = float(eps.replace(SP_VALUE_STR, "").split(":")[1]) - # subsample of x_train is used to avoid reaching of memory limit for large matrices - subsample = list(getattr(x_train, "index", np.arange(x_train.shape[0]))) - np.random.seed(42) - np.random.shuffle(subsample) - subsample = subsample[: min(x_train.shape[0], 1000)] - x_sample = ( - x_train.loc[subsample] if hasattr(x_train, "loc") else x_train[subsample] - ) - # conversion to lower precision is required - # to produce same distances quantile for different dtypes of x - x_sample = x_sample.astype("float32") - dist = np.tril(euclidean_distances(x_sample, x_sample)).reshape(-1) - dist = dist[dist != 0] - quantile = float(np.quantile(dist, quantile)) - set_bench_case_value(bench_case, "algorithm:estimator_params:eps", quantile) + param_value = get_bench_case_value(bench_case, param_name, None) + if is_special_value(param_value) and param_value.replace( + SP_VALUE_STR, "" + ).startswith("distances_quantile"): + set_bench_case_value( + bench_case, param_name, get_distances_quantile(data[0], param_value) + )