From 437c60773686173288283c28d53256378854a9a2 Mon Sep 17 00:00:00 2001 From: Alexandr-Solovev Date: Fri, 18 Sep 2026 08:16:42 -0700 Subject: [PATCH 1/9] feat(configs): add HDBSCAN benchmark configs Add sklearnex-vs-sklearn HDBSCAN CPU benchmarks: * configs/common/hdbscan.json - reusable parameter sets (methods, metrics, density/cluster-selection/store_centers sweeps, dtypes, data formats), modelled on the DBSCAN common config * configs/regular/hdbscan.json - tree- and brute-friendly datasets * configs/experiments/hdbscan_parameters.json - parameter space sweeps * configs/experiments/hdbscan_scaling.json - thread/NUMA core scaling plus n_samples and n_features sweeps Two harness fixes are needed to make these runnable: * enrich_result: normalize a `*.preview.*` library to its base library so preview-namespace estimators (sklearnex.preview.cluster.HDBSCAN) are comparable with the stock implementation in the report * box_filter: fall back to the whole series when the 20-80% box is empty, which happens for 2 measurements - a `time_limit` early stop can leave exactly 2 and the aggregated time became NaN --- configs/common/hdbscan.json | 94 +++++++++ configs/experiments/hdbscan_parameters.json | 129 ++++++++++++ configs/experiments/hdbscan_scaling.json | 217 ++++++++++++++++++++ configs/regular/hdbscan.json | 83 ++++++++ sklbench/benchmarks/common.py | 22 +- sklbench/utils/measurement.py | 5 + 6 files changed, 543 insertions(+), 7 deletions(-) create mode 100644 configs/common/hdbscan.json create mode 100644 configs/experiments/hdbscan_parameters.json create mode 100644 configs/experiments/hdbscan_scaling.json create mode 100644 configs/regular/hdbscan.json diff --git a/configs/common/hdbscan.json b/configs/common/hdbscan.json new file mode 100644 index 00000000..fae556da --- /dev/null +++ b/configs/common/hdbscan.json @@ -0,0 +1,94 @@ +{ + "PARAMETERS_SETS": { + "hdbscan sklearn-ex[cpu] implementations": { + "algorithm": [ + { "library": "sklearn", "device": "cpu" }, + { "library": "sklearnex.preview.cluster", "device": "cpu" } + ] + }, + "common hdbscan parameters": { + "algorithm": { + "estimator": "HDBSCAN", + "estimator_params": { + "min_cluster_size": 5, + "min_samples": 5, + "metric": "euclidean", + "cluster_selection_method": "eom", + "allow_single_cluster": false, + "store_centers": null, + "copy": false + }, + "estimator_methods": { "training": "fit" } + }, + "data": { "format": "numpy", "order": "C", "dtype": "float64" }, + "bench": { "n_runs": 3, "time_limit": 1200 } + }, + "sklearn hdbscan parameters": { + "algorithm": { + "estimator_params": { "n_jobs": "[SPECIAL_VALUE]physical_cpus" } + } + }, + "hdbscan brute method": { + "algorithm": { "estimator_params": { "algorithm": "brute" } } + }, + "hdbscan kd_tree method": { + "algorithm": { "estimator_params": { "algorithm": "kd_tree", "leaf_size": 40 } } + }, + "hdbscan ball_tree method": { + "algorithm": { "estimator_params": { "algorithm": "ball_tree", "leaf_size": 40 } } + }, + "hdbscan tree methods": { + "algorithm": { + "estimator_params": { "algorithm": ["kd_tree", "ball_tree"], "leaf_size": 40 } + } + }, + "hdbscan all methods": { + "algorithm": { + "estimator_params": { + "algorithm": ["brute", "kd_tree", "ball_tree"], + "leaf_size": 40 + } + } + }, + "hdbscan tree metrics": { + "algorithm": { + "estimator_params": { "metric": ["euclidean", "manhattan", "chebyshev"] } + } + }, + "hdbscan minkowski metric": { + "algorithm": { + "estimator_params": { "metric": "minkowski", "metric_params": { "p": 3 } } + } + }, + "hdbscan cosine metric": { + "algorithm": { + "estimator_params": { "metric": "cosine", "algorithm": "brute" } + } + }, + "hdbscan cluster selection sweep": { + "algorithm": { + "estimator_params": { "cluster_selection_method": ["eom", "leaf"] } + } + }, + "hdbscan density sweep": { + "algorithm": { + "estimator_params": { "min_cluster_size": [5, 25, 100], "min_samples": [5, 25] } + } + }, + "hdbscan store centers sweep": { + "algorithm": { + "estimator_params": { "store_centers": [null, "centroid", "medoid", "both"] } + } + }, + "hdbscan dtype sweep": { + "data": { "dtype": ["float32", "float64"] } + }, + "hdbscan data format sweep": { + "data": [ + { "format": "numpy", "order": "C" }, + { "format": "numpy", "order": "F" }, + { "format": "pandas", "order": "F" } + ] + } + } +} diff --git a/configs/experiments/hdbscan_parameters.json b/configs/experiments/hdbscan_parameters.json new file mode 100644 index 00000000..688d855d --- /dev/null +++ b/configs/experiments/hdbscan_parameters.json @@ -0,0 +1,129 @@ +{ + "INCLUDE": ["../common/hdbscan.json"], + "PARAMETERS_SETS": { + "hdbscan parameters data [tree]": { + "data": { + "source": "make_blobs", + "generation_kwargs": { + "centers": 10, + "n_samples": 100000, + "n_features": 8, + "cluster_std": 2.0, + "random_state": 42 + }, + "split_kwargs": { "ignore": true } + } + }, + "hdbscan parameters data [brute]": { + "data": { + "source": "make_blobs", + "generation_kwargs": { + "centers": 10, + "n_samples": 25000, + "n_features": 64, + "cluster_std": 4.0, + "random_state": 42 + }, + "split_kwargs": { "ignore": true } + } + }, + "hdbscan parameters data [real]": { + "data": { + "dataset": "skin_segmentation", + "split_kwargs": { "train_size": 30000 }, + "preprocessing_kwargs": { "normalize": "standard" } + } + } + }, + "TEMPLATES": { + "hdbscan metrics [tree]": { + "SETS": [ + "hdbscan sklearn-ex[cpu] implementations", + "common hdbscan parameters", + "sklearn hdbscan parameters", + "hdbscan tree methods", + "hdbscan tree metrics", + "hdbscan parameters data [tree]" + ] + }, + "hdbscan minkowski [tree]": { + "SETS": [ + "hdbscan sklearn-ex[cpu] implementations", + "common hdbscan parameters", + "sklearn hdbscan parameters", + "hdbscan kd_tree method", + "hdbscan minkowski metric", + "hdbscan parameters data [tree]" + ] + }, + "hdbscan metrics [brute]": { + "SETS": [ + "hdbscan sklearn-ex[cpu] implementations", + "common hdbscan parameters", + "sklearn hdbscan parameters", + "hdbscan brute method", + "hdbscan tree metrics", + "hdbscan parameters data [brute]" + ] + }, + "hdbscan cosine [brute]": { + "SETS": [ + "hdbscan sklearn-ex[cpu] implementations", + "common hdbscan parameters", + "sklearn hdbscan parameters", + "hdbscan cosine metric", + "hdbscan parameters data [brute]" + ] + }, + "hdbscan cluster selection": { + "SETS": [ + "hdbscan sklearn-ex[cpu] implementations", + "common hdbscan parameters", + "sklearn hdbscan parameters", + "hdbscan kd_tree method", + "hdbscan cluster selection sweep", + "hdbscan parameters data [tree]" + ] + }, + "hdbscan density": { + "SETS": [ + "hdbscan sklearn-ex[cpu] implementations", + "common hdbscan parameters", + "sklearn hdbscan parameters", + "hdbscan kd_tree method", + "hdbscan density sweep", + "hdbscan parameters data [tree]" + ] + }, + "hdbscan store centers": { + "SETS": [ + "hdbscan sklearn-ex[cpu] implementations", + "common hdbscan parameters", + "sklearn hdbscan parameters", + "hdbscan kd_tree method", + "hdbscan store centers sweep", + "hdbscan parameters data [tree]" + ] + }, + "hdbscan dtypes": { + "SETS": [ + "hdbscan sklearn-ex[cpu] implementations", + "common hdbscan parameters", + "sklearn hdbscan parameters", + "hdbscan all methods", + "hdbscan dtype sweep", + "hdbscan parameters data [real]" + ] + }, + "hdbscan data formats": { + "SETS": [ + "hdbscan sklearn-ex[cpu] implementations", + "common hdbscan parameters", + "sklearn hdbscan parameters", + "hdbscan kd_tree method", + "hdbscan data format sweep", + "hdbscan parameters data [tree]" + ] + } + } +} diff --git a/configs/experiments/hdbscan_scaling.json b/configs/experiments/hdbscan_scaling.json new file mode 100644 index 00000000..b1c48c3c --- /dev/null +++ b/configs/experiments/hdbscan_scaling.json @@ -0,0 +1,217 @@ +{ + "INCLUDE": ["../common/hdbscan.json"], + "PARAMETERS_SETS": { + "sklearnex hdbscan implementation": { + "algorithm": { "library": "sklearnex.preview.cluster", "device": "cpu" } + }, + "hdbscan thread scaling": [ + { + "bench": { "taskset": "0" }, + "algorithm": { "estimator_params": { "n_jobs": 1 } } + }, + { + "bench": { "taskset": "0-3" }, + "algorithm": { "estimator_params": { "n_jobs": 4 } } + }, + { + "bench": { "taskset": "0-7" }, + "algorithm": { "estimator_params": { "n_jobs": 8 } } + }, + { + "bench": { "taskset": "0-15" }, + "algorithm": { "estimator_params": { "n_jobs": 16 } } + }, + { + "bench": { "taskset": "0-27" }, + "algorithm": { "estimator_params": { "n_jobs": 28 } } + }, + { + "bench": { "taskset": "0-55" }, + "algorithm": { "estimator_params": { "n_jobs": 56 } } + }, + { + "bench": { "taskset": "0-111" }, + "algorithm": { "estimator_params": { "n_jobs": 112 } } + }, + { + "bench": { "taskset": "0-223" }, + "algorithm": { "estimator_params": { "n_jobs": 224 } } + } + ], + "hdbscan numa scaling": [ + { + "bench": { "taskset": "0-55" }, + "algorithm": { "estimator_params": { "n_jobs": 56 } } + }, + { + "bench": { "taskset": "0-27,56-83" }, + "algorithm": { "estimator_params": { "n_jobs": 56 } } + }, + { + "bench": { "taskset": "0-111" }, + "algorithm": { "estimator_params": { "n_jobs": 112 } } + } + ], + "hdbscan scaling profiling": { + "bench": { + "n_runs": 3, + "time_limit": 1200, + "cpu_profile": true, + "memory_profile": true + } + }, + "hdbscan brute scaling data": { + "data": { + "source": "make_blobs", + "generation_kwargs": { + "centers": 10, + "n_samples": 50000, + "n_features": 256, + "cluster_std": 4.0, + "random_state": 42 + }, + "split_kwargs": { "ignore": true } + } + }, + "hdbscan tree scaling data": { + "data": { + "source": "make_blobs", + "generation_kwargs": { + "centers": 10, + "n_samples": 1000000, + "n_features": 16, + "cluster_std": 2.0, + "random_state": 42 + }, + "split_kwargs": { "ignore": true } + } + }, + "hdbscan n_samples sweep [brute]": { + "data": { + "source": "make_blobs", + "generation_kwargs": { + "centers": 10, + "n_samples": "[RANGE]mul:5000:40000:2", + "n_features": 64, + "cluster_std": 4.0, + "random_state": 42 + }, + "split_kwargs": { "ignore": true } + } + }, + "hdbscan n_samples sweep [tree]": { + "data": { + "source": "make_blobs", + "generation_kwargs": { + "centers": 10, + "n_samples": "[RANGE]mul:25000:400000:2", + "n_features": 8, + "cluster_std": 2.0, + "random_state": 42 + }, + "split_kwargs": { "ignore": true } + } + }, + "hdbscan n_samples sweep [tree, sklearnex only]": { + "data": { + "source": "make_blobs", + "generation_kwargs": { + "centers": 10, + "n_samples": "[RANGE]mul:500000:8000000:2", + "n_features": 8, + "cluster_std": 2.0, + "random_state": 42 + }, + "split_kwargs": { "ignore": true } + } + }, + "hdbscan n_features sweep": { + "data": { + "source": "make_blobs", + "generation_kwargs": { + "centers": 10, + "n_samples": 20000, + "n_features": "[RANGE]pow:2:2:10", + "cluster_std": 4.0, + "random_state": 42 + }, + "split_kwargs": { "ignore": true } + } + }, + "hdbscan size sweep limits": { + "bench": { "n_runs": 3, "time_limit": 300 } + } + }, + "TEMPLATES": { + "hdbscan thread scaling [brute]": { + "SETS": [ + "sklearnex hdbscan implementation", + "common hdbscan parameters", + "hdbscan scaling profiling", + "hdbscan brute method", + "hdbscan thread scaling", + "hdbscan brute scaling data" + ] + }, + "hdbscan thread scaling [kd_tree]": { + "SETS": [ + "sklearnex hdbscan implementation", + "common hdbscan parameters", + "hdbscan scaling profiling", + "hdbscan kd_tree method", + "hdbscan thread scaling", + "hdbscan tree scaling data" + ] + }, + "hdbscan numa scaling [brute]": { + "SETS": [ + "sklearnex hdbscan implementation", + "common hdbscan parameters", + "hdbscan scaling profiling", + "hdbscan brute method", + "hdbscan numa scaling", + "hdbscan brute scaling data" + ] + }, + "hdbscan n_samples scaling [brute]": { + "SETS": [ + "hdbscan sklearn-ex[cpu] implementations", + "common hdbscan parameters", + "sklearn hdbscan parameters", + "hdbscan brute method", + "hdbscan n_samples sweep [brute]", + "hdbscan size sweep limits" + ] + }, + "hdbscan n_samples scaling [kd_tree]": { + "SETS": [ + "hdbscan sklearn-ex[cpu] implementations", + "common hdbscan parameters", + "sklearn hdbscan parameters", + "hdbscan kd_tree method", + "hdbscan n_samples sweep [tree]", + "hdbscan size sweep limits" + ] + }, + "hdbscan n_samples scaling [kd_tree, beyond sklearn]": { + "SETS": [ + "sklearnex hdbscan implementation", + "common hdbscan parameters", + "sklearn hdbscan parameters", + "hdbscan kd_tree method", + "hdbscan n_samples sweep [tree, sklearnex only]", + "hdbscan size sweep limits" + ] + }, + "hdbscan n_features scaling": { + "SETS": [ + "hdbscan sklearn-ex[cpu] implementations", + "common hdbscan parameters", + "sklearn hdbscan parameters", + "hdbscan all methods", + "hdbscan n_features sweep", + "hdbscan size sweep limits" + ] + } + } +} diff --git a/configs/regular/hdbscan.json b/configs/regular/hdbscan.json new file mode 100644 index 00000000..29431fd7 --- /dev/null +++ b/configs/regular/hdbscan.json @@ -0,0 +1,83 @@ +{ + "INCLUDE": ["../common/sklearn.json", "../common/hdbscan.json"], + "PARAMETERS_SETS": { + "hdbscan tree-friendly datasets": { + "data": [ + { + "dataset": "skin_segmentation", + "split_kwargs": { "train_size": 100000 }, + "preprocessing_kwargs": { "normalize": "standard" } + }, + { + "dataset": "road_network", + "split_kwargs": { "train_size": 100000 }, + "preprocessing_kwargs": { "normalize": "standard" } + }, + { + "dataset": "covtype", + "split_kwargs": { "train_size": 50000 }, + "preprocessing_kwargs": { "normalize": "standard" } + }, + { + "source": "make_blobs", + "generation_kwargs": { + "centers": 10, + "n_samples": 100000, + "n_features": [4, 16, 64], + "cluster_std": 2.0 + }, + "split_kwargs": { "ignore": true } + } + ] + }, + "hdbscan brute-friendly datasets": { + "data": [ + { + "dataset": "mnist", + "split_kwargs": { "train_size": 20000 }, + "preprocessing_kwargs": { "normalize": "standard" } + }, + { + "dataset": "cifar", + "split_kwargs": { "train_size": 15000 }, + "preprocessing_kwargs": { "normalize": "mean" } + }, + { + "dataset": "sensit", + "split_kwargs": { "train_size": 25000 }, + "preprocessing_kwargs": { "normalize": "standard" } + }, + { + "source": "make_blobs", + "generation_kwargs": { + "centers": 10, + "n_samples": 25000, + "n_features": [64, 256, 1024], + "cluster_std": 4.0 + }, + "split_kwargs": { "ignore": true } + } + ] + } + }, + "TEMPLATES": { + "hdbscan tree methods": { + "SETS": [ + "hdbscan sklearn-ex[cpu] implementations", + "common hdbscan parameters", + "sklearn hdbscan parameters", + "hdbscan tree methods", + "hdbscan tree-friendly datasets" + ] + }, + "hdbscan brute force": { + "SETS": [ + "hdbscan sklearn-ex[cpu] implementations", + "common hdbscan parameters", + "sklearn hdbscan parameters", + "hdbscan brute method", + "hdbscan brute-friendly datasets" + ] + } + } +} diff --git a/sklbench/benchmarks/common.py b/sklbench/benchmarks/common.py index 1df1e1a5..bd7b3376 100644 --- a/sklbench/benchmarks/common.py +++ b/sklbench/benchmarks/common.py @@ -16,6 +16,7 @@ import argparse import json +import re from typing import Dict from ..utils.bench_case import get_bench_case_value, get_data_name @@ -26,16 +27,23 @@ def enrich_result(result: Dict, bench_case: BenchCase) -> Dict: """Common function for all benchmarks to update the result with additional information""" + library = ( + get_bench_case_value(bench_case, "algorithm:library") + .replace( + # skipping emulators namespace for conciseness + "sklbench.emulators.", + "", + ) + .replace(".utils", "") + ) + # estimators available only from a preview namespace (like `sklearnex.preview.cluster`) + # are reported under their library so that they are comparable + # with the stock implementation in the report + library = re.sub(r"\.preview(\..+)?$", "", library) result.update( { "dataset": get_data_name(bench_case, shortened=True), - "library": get_bench_case_value(bench_case, "algorithm:library") - .replace( - # skipping emulators namespace for conciseness - "sklbench.emulators.", - "", - ) - .replace(".utils", ""), + "library": library, "device": get_bench_case_value(bench_case, "algorithm:device"), } ) diff --git a/sklbench/utils/measurement.py b/sklbench/utils/measurement.py index 4df6c57b..66e1f100 100644 --- a/sklbench/utils/measurement.py +++ b/sklbench/utils/measurement.py @@ -56,6 +56,11 @@ def box_filter(array, left=0.2, right=0.8): return array[0], 0.0 lower, upper = array[int(size * left)], array[int(size * right)] result = np.array([item for item in array if lower < item < upper]) + if result.size == 0: + # the box is empty for short series (2 measurements always are, and a + # `time_limit` early stop can leave exactly 2), which would make the + # aggregated time NaN - fall back to the whole series instead + result = np.array(array) return np.mean(result), np.std(result) From f36594729d289b5b8d131a8245be65f991d50086 Mon Sep 17 00:00:00 2001 From: Alexandr-Solovev Date: Tue, 22 Sep 2026 02:40:03 -0700 Subject: [PATCH 2/9] Shorten the regular HDBSCAN config and move the sweeps out of common Review feedback: the regular scope was too heavy and the sweep parameter sets only ever served the experiment configs. * configs/regular/hdbscan.json: 36 -> 8 cases. One tree method (kd_tree) on two datasets and brute force on two, which is enough to track performance changes. * configs/weekly/hdbscan.json: new, 30 cases. Takes over the high-load part of what regular used to run: road_network/covtype/cifar/sensit and the n_features blob sweeps over both tree methods. * configs/common/hdbscan.json: keeps only the sets shared across scopes (implementations, common/sklearn parameters, method selection). The metric, cluster-selection, density, store_centers, dtype and data format sweeps moved into configs/experiments/hdbscan_parameters.json, their only consumer. * configs/experiments/README.md: describe the two HDBSCAN experiments. Co-Authored-By: Claude Opus 5 --- configs/common/hdbscan.json | 40 ------------ configs/experiments/README.md | 4 ++ configs/experiments/hdbscan_parameters.json | 40 ++++++++++++ configs/regular/hdbscan.json | 28 ++------- configs/weekly/hdbscan.json | 68 +++++++++++++++++++++ 5 files changed, 116 insertions(+), 64 deletions(-) create mode 100644 configs/weekly/hdbscan.json diff --git a/configs/common/hdbscan.json b/configs/common/hdbscan.json index fae556da..3b2fe12f 100644 --- a/configs/common/hdbscan.json +++ b/configs/common/hdbscan.json @@ -49,46 +49,6 @@ "leaf_size": 40 } } - }, - "hdbscan tree metrics": { - "algorithm": { - "estimator_params": { "metric": ["euclidean", "manhattan", "chebyshev"] } - } - }, - "hdbscan minkowski metric": { - "algorithm": { - "estimator_params": { "metric": "minkowski", "metric_params": { "p": 3 } } - } - }, - "hdbscan cosine metric": { - "algorithm": { - "estimator_params": { "metric": "cosine", "algorithm": "brute" } - } - }, - "hdbscan cluster selection sweep": { - "algorithm": { - "estimator_params": { "cluster_selection_method": ["eom", "leaf"] } - } - }, - "hdbscan density sweep": { - "algorithm": { - "estimator_params": { "min_cluster_size": [5, 25, 100], "min_samples": [5, 25] } - } - }, - "hdbscan store centers sweep": { - "algorithm": { - "estimator_params": { "store_centers": [null, "centroid", "medoid", "both"] } - } - }, - "hdbscan dtype sweep": { - "data": { "dtype": ["float32", "float64"] } - }, - "hdbscan data format sweep": { - "data": [ - { "format": "numpy", "order": "C" }, - { "format": "numpy", "order": "F" }, - { "format": "pandas", "order": "F" } - ] } } } diff --git a/configs/experiments/README.md b/configs/experiments/README.md index 2b6225c5..dc442c9b 100644 --- a/configs/experiments/README.md +++ b/configs/experiments/README.md @@ -2,4 +2,8 @@ `daal4py_svd`: tests performance scalability of `daal4py.svd` algorithm +`hdbscan_parameters`: sweeps the `HDBSCAN` parameter space (metrics, cluster selection, density thresholds, stored centers, dtypes and data formats) over the `sklearn` and `sklearnex` implementations. + +`hdbscan_scaling`: tests thread, NUMA, `n_samples` and `n_features` scalability of `HDBSCAN`. + `nearest_neighbors`: tests performance of neighbors search implementations from `sklearnex`, `sklearn`, `raft`, `faiss` and `svs`. diff --git a/configs/experiments/hdbscan_parameters.json b/configs/experiments/hdbscan_parameters.json index 688d855d..fe546ec4 100644 --- a/configs/experiments/hdbscan_parameters.json +++ b/configs/experiments/hdbscan_parameters.json @@ -1,6 +1,46 @@ { "INCLUDE": ["../common/hdbscan.json"], "PARAMETERS_SETS": { + "hdbscan tree metrics": { + "algorithm": { + "estimator_params": { "metric": ["euclidean", "manhattan", "chebyshev"] } + } + }, + "hdbscan minkowski metric": { + "algorithm": { + "estimator_params": { "metric": "minkowski", "metric_params": { "p": 3 } } + } + }, + "hdbscan cosine metric": { + "algorithm": { + "estimator_params": { "metric": "cosine", "algorithm": "brute" } + } + }, + "hdbscan cluster selection sweep": { + "algorithm": { + "estimator_params": { "cluster_selection_method": ["eom", "leaf"] } + } + }, + "hdbscan density sweep": { + "algorithm": { + "estimator_params": { "min_cluster_size": [5, 25, 100], "min_samples": [5, 25] } + } + }, + "hdbscan store centers sweep": { + "algorithm": { + "estimator_params": { "store_centers": [null, "centroid", "medoid", "both"] } + } + }, + "hdbscan dtype sweep": { + "data": { "dtype": ["float32", "float64"] } + }, + "hdbscan data format sweep": { + "data": [ + { "format": "numpy", "order": "C" }, + { "format": "numpy", "order": "F" }, + { "format": "pandas", "order": "F" } + ] + }, "hdbscan parameters data [tree]": { "data": { "source": "make_blobs", diff --git a/configs/regular/hdbscan.json b/configs/regular/hdbscan.json index 29431fd7..05c26637 100644 --- a/configs/regular/hdbscan.json +++ b/configs/regular/hdbscan.json @@ -8,22 +8,12 @@ "split_kwargs": { "train_size": 100000 }, "preprocessing_kwargs": { "normalize": "standard" } }, - { - "dataset": "road_network", - "split_kwargs": { "train_size": 100000 }, - "preprocessing_kwargs": { "normalize": "standard" } - }, - { - "dataset": "covtype", - "split_kwargs": { "train_size": 50000 }, - "preprocessing_kwargs": { "normalize": "standard" } - }, { "source": "make_blobs", "generation_kwargs": { "centers": 10, "n_samples": 100000, - "n_features": [4, 16, 64], + "n_features": 16, "cluster_std": 2.0 }, "split_kwargs": { "ignore": true } @@ -37,22 +27,12 @@ "split_kwargs": { "train_size": 20000 }, "preprocessing_kwargs": { "normalize": "standard" } }, - { - "dataset": "cifar", - "split_kwargs": { "train_size": 15000 }, - "preprocessing_kwargs": { "normalize": "mean" } - }, - { - "dataset": "sensit", - "split_kwargs": { "train_size": 25000 }, - "preprocessing_kwargs": { "normalize": "standard" } - }, { "source": "make_blobs", "generation_kwargs": { "centers": 10, "n_samples": 25000, - "n_features": [64, 256, 1024], + "n_features": 256, "cluster_std": 4.0 }, "split_kwargs": { "ignore": true } @@ -61,12 +41,12 @@ } }, "TEMPLATES": { - "hdbscan tree methods": { + "hdbscan kd_tree": { "SETS": [ "hdbscan sklearn-ex[cpu] implementations", "common hdbscan parameters", "sklearn hdbscan parameters", - "hdbscan tree methods", + "hdbscan kd_tree method", "hdbscan tree-friendly datasets" ] }, diff --git a/configs/weekly/hdbscan.json b/configs/weekly/hdbscan.json new file mode 100644 index 00000000..a11d82a2 --- /dev/null +++ b/configs/weekly/hdbscan.json @@ -0,0 +1,68 @@ +{ + "INCLUDE": ["../common/sklearn.json", "../common/hdbscan.json"], + "PARAMETERS_SETS": { + "high-load hdbscan tree-friendly datasets": { + "data": [ + { + "dataset": ["road_network", "covtype"], + "split_kwargs": { "train_size": 100000 }, + "preprocessing_kwargs": { "normalize": "standard" } + }, + { + "source": "make_blobs", + "generation_kwargs": { + "centers": 10, + "n_samples": 100000, + "n_features": [4, 16, 64], + "cluster_std": 2.0 + }, + "split_kwargs": { "ignore": true } + } + ] + }, + "high-load hdbscan brute-friendly datasets": { + "data": [ + { + "dataset": "cifar", + "split_kwargs": { "train_size": 15000 }, + "preprocessing_kwargs": { "normalize": "mean" } + }, + { + "dataset": "sensit", + "split_kwargs": { "train_size": 25000 }, + "preprocessing_kwargs": { "normalize": "standard" } + }, + { + "source": "make_blobs", + "generation_kwargs": { + "centers": 10, + "n_samples": 25000, + "n_features": [64, 256, 1024], + "cluster_std": 4.0 + }, + "split_kwargs": { "ignore": true } + } + ] + } + }, + "TEMPLATES": { + "hdbscan tree methods": { + "SETS": [ + "hdbscan sklearn-ex[cpu] implementations", + "common hdbscan parameters", + "sklearn hdbscan parameters", + "hdbscan tree methods", + "high-load hdbscan tree-friendly datasets" + ] + }, + "hdbscan brute force": { + "SETS": [ + "hdbscan sklearn-ex[cpu] implementations", + "common hdbscan parameters", + "sklearn hdbscan parameters", + "hdbscan brute method", + "high-load hdbscan brute-friendly datasets" + ] + } + } +} From 530468e0c846e44dc33a5d535132da5dadcbe8c7 Mon Sep 17 00:00:00 2001 From: Alexandr-Solovev Date: Tue, 22 Sep 2026 07:20:44 -0700 Subject: [PATCH 3/9] Spread the regular HDBSCAN cases over distinct data situations The four datasets in the regular config were picked for size, which left two near-duplicate pairs. Each now covers a different regime: * kd_tree, skin_segmentation 100000 x 3 -- real, discrete, duplicate-heavy, so low dimension with thousands of small clusters and a tie-dense mutual reachability graph. * kd_tree, make_blobs 100000 x 8 with 100 centers (was 16 features, 10 centers) -- many well-separated clusters, which moves the cost into the condensed tree and the cluster selection instead of the neighbors search. Also no longer a subset of the weekly sweep, which covers 10 centers at 4 / 16 / 64 features. * brute, mnist 20000 x 784 -- high dimension, where the pairwise distance computation dominates. * brute, make_blobs 25000 x 4 (was 256 features) -- the opposite brute-force regime: the distance computation is negligible and the passes over the n^2 matrix dominate. Still 8 cases. Verified end to end against stock scikit-learn: 345.8x, 176.7x, 5.2x and 5.7x respectively, with matching cluster counts and Davies-Bouldin scores. Co-Authored-By: Claude Opus 5 --- configs/regular/hdbscan.json | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/configs/regular/hdbscan.json b/configs/regular/hdbscan.json index 05c26637..e96fb314 100644 --- a/configs/regular/hdbscan.json +++ b/configs/regular/hdbscan.json @@ -11,10 +11,10 @@ { "source": "make_blobs", "generation_kwargs": { - "centers": 10, + "centers": 100, "n_samples": 100000, - "n_features": 16, - "cluster_std": 2.0 + "n_features": 8, + "cluster_std": 1.0 }, "split_kwargs": { "ignore": true } } @@ -32,8 +32,8 @@ "generation_kwargs": { "centers": 10, "n_samples": 25000, - "n_features": 256, - "cluster_std": 4.0 + "n_features": 4, + "cluster_std": 1.0 }, "split_kwargs": { "ignore": true } } From b18905fdaeefbc8f82d7918ea3da4742ef03a77d Mon Sep 17 00:00:00 2001 From: Alexandr-Solovev Date: Mon, 5 Oct 2026 04:00:54 -0700 Subject: [PATCH 4/9] configs(hdbscan): point at the released sklearnex HDBSCAN HDBSCAN left sklearnex preview, so `sklearnex.preview.cluster` no longer provides it. With the plain `sklearnex` library the hdbscan-specific implementation sets are identical to the shared ones in `common/sklearn.json`, so they are dropped and the configs use the shared sets directly. The regular config gains the GPU implementation the DBSCAN one already has. The implementation set is listed last in the templates that include GPU: it carries the `dpnp` data format, which `common hdbscan parameters` would otherwise overwrite with `numpy`. Co-Authored-By: Claude Opus 5 --- configs/common/hdbscan.json | 6 ------ configs/experiments/hdbscan_parameters.json | 20 ++++++++++---------- configs/experiments/hdbscan_scaling.json | 10 +++++----- configs/regular/hdbscan.json | 8 ++++---- configs/weekly/hdbscan.json | 4 ++-- 5 files changed, 21 insertions(+), 27 deletions(-) diff --git a/configs/common/hdbscan.json b/configs/common/hdbscan.json index 3b2fe12f..fd82b70d 100644 --- a/configs/common/hdbscan.json +++ b/configs/common/hdbscan.json @@ -1,11 +1,5 @@ { "PARAMETERS_SETS": { - "hdbscan sklearn-ex[cpu] implementations": { - "algorithm": [ - { "library": "sklearn", "device": "cpu" }, - { "library": "sklearnex.preview.cluster", "device": "cpu" } - ] - }, "common hdbscan parameters": { "algorithm": { "estimator": "HDBSCAN", diff --git a/configs/experiments/hdbscan_parameters.json b/configs/experiments/hdbscan_parameters.json index fe546ec4..8e1f57ee 100644 --- a/configs/experiments/hdbscan_parameters.json +++ b/configs/experiments/hdbscan_parameters.json @@ -1,5 +1,5 @@ { - "INCLUDE": ["../common/hdbscan.json"], + "INCLUDE": ["../common/sklearn.json", "../common/hdbscan.json"], "PARAMETERS_SETS": { "hdbscan tree metrics": { "algorithm": { @@ -78,7 +78,7 @@ "TEMPLATES": { "hdbscan metrics [tree]": { "SETS": [ - "hdbscan sklearn-ex[cpu] implementations", + "sklearn-ex[cpu] implementations", "common hdbscan parameters", "sklearn hdbscan parameters", "hdbscan tree methods", @@ -88,7 +88,7 @@ }, "hdbscan minkowski [tree]": { "SETS": [ - "hdbscan sklearn-ex[cpu] implementations", + "sklearn-ex[cpu] implementations", "common hdbscan parameters", "sklearn hdbscan parameters", "hdbscan kd_tree method", @@ -98,7 +98,7 @@ }, "hdbscan metrics [brute]": { "SETS": [ - "hdbscan sklearn-ex[cpu] implementations", + "sklearn-ex[cpu] implementations", "common hdbscan parameters", "sklearn hdbscan parameters", "hdbscan brute method", @@ -108,7 +108,7 @@ }, "hdbscan cosine [brute]": { "SETS": [ - "hdbscan sklearn-ex[cpu] implementations", + "sklearn-ex[cpu] implementations", "common hdbscan parameters", "sklearn hdbscan parameters", "hdbscan cosine metric", @@ -117,7 +117,7 @@ }, "hdbscan cluster selection": { "SETS": [ - "hdbscan sklearn-ex[cpu] implementations", + "sklearn-ex[cpu] implementations", "common hdbscan parameters", "sklearn hdbscan parameters", "hdbscan kd_tree method", @@ -127,7 +127,7 @@ }, "hdbscan density": { "SETS": [ - "hdbscan sklearn-ex[cpu] implementations", + "sklearn-ex[cpu] implementations", "common hdbscan parameters", "sklearn hdbscan parameters", "hdbscan kd_tree method", @@ -137,7 +137,7 @@ }, "hdbscan store centers": { "SETS": [ - "hdbscan sklearn-ex[cpu] implementations", + "sklearn-ex[cpu] implementations", "common hdbscan parameters", "sklearn hdbscan parameters", "hdbscan kd_tree method", @@ -147,7 +147,7 @@ }, "hdbscan dtypes": { "SETS": [ - "hdbscan sklearn-ex[cpu] implementations", + "sklearn-ex[cpu] implementations", "common hdbscan parameters", "sklearn hdbscan parameters", "hdbscan all methods", @@ -157,7 +157,7 @@ }, "hdbscan data formats": { "SETS": [ - "hdbscan sklearn-ex[cpu] implementations", + "sklearn-ex[cpu] implementations", "common hdbscan parameters", "sklearn hdbscan parameters", "hdbscan kd_tree method", diff --git a/configs/experiments/hdbscan_scaling.json b/configs/experiments/hdbscan_scaling.json index b1c48c3c..997609a6 100644 --- a/configs/experiments/hdbscan_scaling.json +++ b/configs/experiments/hdbscan_scaling.json @@ -1,8 +1,8 @@ { - "INCLUDE": ["../common/hdbscan.json"], + "INCLUDE": ["../common/sklearn.json", "../common/hdbscan.json"], "PARAMETERS_SETS": { "sklearnex hdbscan implementation": { - "algorithm": { "library": "sklearnex.preview.cluster", "device": "cpu" } + "algorithm": { "library": "sklearnex", "device": "cpu" } }, "hdbscan thread scaling": [ { @@ -175,7 +175,7 @@ }, "hdbscan n_samples scaling [brute]": { "SETS": [ - "hdbscan sklearn-ex[cpu] implementations", + "sklearn-ex[cpu] implementations", "common hdbscan parameters", "sklearn hdbscan parameters", "hdbscan brute method", @@ -185,7 +185,7 @@ }, "hdbscan n_samples scaling [kd_tree]": { "SETS": [ - "hdbscan sklearn-ex[cpu] implementations", + "sklearn-ex[cpu] implementations", "common hdbscan parameters", "sklearn hdbscan parameters", "hdbscan kd_tree method", @@ -205,7 +205,7 @@ }, "hdbscan n_features scaling": { "SETS": [ - "hdbscan sklearn-ex[cpu] implementations", + "sklearn-ex[cpu] implementations", "common hdbscan parameters", "sklearn hdbscan parameters", "hdbscan all methods", diff --git a/configs/regular/hdbscan.json b/configs/regular/hdbscan.json index e96fb314..1f2bb69f 100644 --- a/configs/regular/hdbscan.json +++ b/configs/regular/hdbscan.json @@ -43,20 +43,20 @@ "TEMPLATES": { "hdbscan kd_tree": { "SETS": [ - "hdbscan sklearn-ex[cpu] implementations", "common hdbscan parameters", "sklearn hdbscan parameters", "hdbscan kd_tree method", - "hdbscan tree-friendly datasets" + "hdbscan tree-friendly datasets", + "sklearn-ex[cpu,gpu] implementations" ] }, "hdbscan brute force": { "SETS": [ - "hdbscan sklearn-ex[cpu] implementations", "common hdbscan parameters", "sklearn hdbscan parameters", "hdbscan brute method", - "hdbscan brute-friendly datasets" + "hdbscan brute-friendly datasets", + "sklearn-ex[cpu,gpu] implementations" ] } } diff --git a/configs/weekly/hdbscan.json b/configs/weekly/hdbscan.json index a11d82a2..0c9dac0f 100644 --- a/configs/weekly/hdbscan.json +++ b/configs/weekly/hdbscan.json @@ -48,7 +48,7 @@ "TEMPLATES": { "hdbscan tree methods": { "SETS": [ - "hdbscan sklearn-ex[cpu] implementations", + "sklearn-ex[cpu] implementations", "common hdbscan parameters", "sklearn hdbscan parameters", "hdbscan tree methods", @@ -57,7 +57,7 @@ }, "hdbscan brute force": { "SETS": [ - "hdbscan sklearn-ex[cpu] implementations", + "sklearn-ex[cpu] implementations", "common hdbscan parameters", "sklearn hdbscan parameters", "hdbscan brute method", From d4b73664227f86d885c6b36c4b147d66cdebbf92 Mon Sep 17 00:00:00 2001 From: Alexandr-Solovev Date: Mon, 5 Oct 2026 04:01:07 -0700 Subject: [PATCH 5/9] feat: measure estimator methods that take their own parameters A method listed in the new `algorithm:method_params:{method}` is called with those parameters and no data, which is what methods working off the fitted model need. `HDBSCAN.dbscan_clustering` is the case at hand: it re-cuts the hierarchy `fit` already built, at a distance of its own. `cut_distance` of that method also accepts the `distances_quantile` special value, so an HDBSCAN hierarchy can be re-cut at exactly the distance DBSCAN is measured at on the same data. The quantile computation moves into `get_distances_quantile` and is shared with `eps`. Co-Authored-By: Claude Opus 5 --- configs/BENCH-CONFIG-SPEC.md | 2 + sklbench/benchmarks/sklearn_estimator.py | 14 +++++- sklbench/utils/special_params.py | 60 +++++++++++++++--------- 3 files changed, 53 insertions(+), 23 deletions(-) diff --git a/configs/BENCH-CONFIG-SPEC.md b/configs/BENCH-CONFIG-SPEC.md index 26bff24f..0c2a7b56 100644 --- a/configs/BENCH-CONFIG-SPEC.md +++ b/configs/BENCH-CONFIG-SPEC.md @@ -108,6 +108,7 @@ Configs have the three highest parameter keys: |:---------------|:--------------|:--------|:------------| | `algorithm`:`estimator` | None | | Name of measured estimator. | | `algorithm`:`estimator_params` | Empty `dict` | | Parameters for estimator constructor. | +| `algorithm`:`method_params`:`{method}` | None | | Parameters for a measured method of estimator. A method listed here is called with these parameters only, without data, which fits methods working off the fitted model such as `HDBSCAN.dbscan_clustering`. | | `algorithm`:`batch_size`:`{stage}` | None | Any positive integer | Enables online mode for `{stage}` methods of estimator (sequential calls for each batch). | | `algorithm`:`sklearn_context` | None | | Parameters for sklearn `config_context` used over estimator. `array_api_dispatch` requires `SCIPY_ARRAY_API=1`, which scikit-learn_bench sets by default if it is unset in the environment. | | `algorithm`:`sklearnex_context` | None | | Parameters for sklearnex `config_context` used over estimator. Updated by `sklearn_context` if set. | @@ -140,6 +141,7 @@ List of available special values: | `algorithm`:`estimator_params`:`scale_pos_weight` | sklearn_estimator | `auto` | Sets `scale_pos_weight` parameter to `sum(negative instances) / sum(positive instances)` value for estimator. | | `algorithm`:`estimator_params`:`n_clusters` | sklearn_estimator | `auto` | Sets `n_clusters` parameter to number of clusters or classes from dataset description for estimator. | | `algorithm`:`estimator_params`:`eps` | sklearn_estimator | `distances_quantile:{quantile}` format where quantile is *float* value in [0, 1] range | Computes `eps` parameter as quantile value of distances in `x_train` matrix for estimator. | +| `algorithm`:`method_params`:`dbscan_clustering`:`cut_distance` | sklearn_estimator | `distances_quantile:{quantile}` format where quantile is *float* value in [0, 1] range | Computes `cut_distance` the same way as `eps` above, so that an HDBSCAN hierarchy can be re-cut at the distance DBSCAN is measured at. | ## Range of Values diff --git a/sklbench/benchmarks/sklearn_estimator.py b/sklbench/benchmarks/sklearn_estimator.py index 8c511ce3..99fbda1e 100644 --- a/sklbench/benchmarks/sklearn_estimator.py +++ b/sklbench/benchmarks/sklearn_estimator.py @@ -449,7 +449,15 @@ def measure_sklearn_estimator( for method in estimator_methods[stage]: if hasattr(estimator_instance, method): method_instance = getattr(estimator_instance, method) - if "y" in list(inspect.signature(method_instance).parameters): + # a method given its own arguments takes no data: it works off the + # fitted model instead, like 'HDBSCAN.dbscan_clustering', which + # re-cuts the hierarchy 'fit' built at another distance + method_params = get_bench_case_value( + bench_case, f"algorithm:method_params:{method}", None + ) + if method_params is not None: + data_args = () + elif "y" in list(inspect.signature(method_instance).parameters): if stage == "training": data_args = (x_train, y_train) else: @@ -484,7 +492,9 @@ def measure_sklearn_estimator( ) method_instance = getattr(daal_model, method) - metrics[method] = measure_case(bench_case, method_instance, *data_args) + metrics[method] = measure_case( + bench_case, method_instance, *data_args, **(method_params or {}) + ) if ensure_sklearnex_patching: full_method_name = f"{estimator_class.__name__}.{method}" sklearnex_logging_stream.seek(0) diff --git a/sklbench/utils/special_params.py b/sklbench/utils/special_params.py index 80812012..95b0bceb 100644 --- a/sklbench/utils/special_params.py +++ b/sklbench/utils/special_params.py @@ -148,6 +148,32 @@ def get_ratio_from_n_jobs(n_jobs: str) -> float: raise ValueError(f'Wrong arguments {args} in "n_jobs" special value') +def get_distances_quantile(x_train, special_value: str) -> float: + """Computes a quantile of the pairwise euclidean distances of `x_train`. + + Args: + x_train: Training data in any of the supported formats. + special_value: Special value of the form `distances_quantile:{quantile}`. + + Returns: + The requested quantile of the non-zero pairwise distances. + """ + x_train = convert_to_numpy(x_train) + quantile = float(special_value.replace(SP_VALUE_STR, "").split(":")[1]) + # subsample of x_train is used to avoid reaching of memory limit for large matrices + subsample = list(getattr(x_train, "index", np.arange(x_train.shape[0]))) + np.random.seed(42) + np.random.shuffle(subsample) + subsample = subsample[: min(x_train.shape[0], 1000)] + x_sample = x_train.loc[subsample] if hasattr(x_train, "loc") else x_train[subsample] + # conversion to lower precision is required + # to produce same distances quantile for different dtypes of x + x_sample = x_sample.astype("float32") + dist = np.tril(euclidean_distances(x_sample, x_sample)).reshape(-1) + dist = dist[dist != 0] + return float(np.quantile(dist, quantile)) + + def assign_case_special_values_on_run( bench_case: BenchCase, data, data_description: Dict ): @@ -270,25 +296,17 @@ def assign_case_special_values_on_run( "Unable to auto-assign n_clusters: " "data description doesn't have n_clusters or n_classes" ) - # "eps" auto assignment for DBSCAN - eps = get_bench_case_value(bench_case, "algorithm:estimator_params:eps", None) - if is_special_value(eps) and eps.replace(SP_VALUE_STR, "").startswith( - "distances_quantile" + # "eps" auto assignment for DBSCAN, and the equivalent "cut_distance" for the + # re-cut of an HDBSCAN hierarchy, so that both are the same distance on the + # same data and the two clusterings can be compared + for param_name in ( + "algorithm:estimator_params:eps", + "algorithm:method_params:dbscan_clustering:cut_distance", ): - x_train = convert_to_numpy(data[0]) - quantile = float(eps.replace(SP_VALUE_STR, "").split(":")[1]) - # subsample of x_train is used to avoid reaching of memory limit for large matrices - subsample = list(getattr(x_train, "index", np.arange(x_train.shape[0]))) - np.random.seed(42) - np.random.shuffle(subsample) - subsample = subsample[: min(x_train.shape[0], 1000)] - x_sample = ( - x_train.loc[subsample] if hasattr(x_train, "loc") else x_train[subsample] - ) - # conversion to lower precision is required - # to produce same distances quantile for different dtypes of x - x_sample = x_sample.astype("float32") - dist = np.tril(euclidean_distances(x_sample, x_sample)).reshape(-1) - dist = dist[dist != 0] - quantile = float(np.quantile(dist, quantile)) - set_bench_case_value(bench_case, "algorithm:estimator_params:eps", quantile) + param_value = get_bench_case_value(bench_case, param_name, None) + if is_special_value(param_value) and param_value.replace( + SP_VALUE_STR, "" + ).startswith("distances_quantile"): + set_bench_case_value( + bench_case, param_name, get_distances_quantile(data[0], param_value) + ) From 7994a317936b8f4aa930d7aab1aff082f78f17b1 Mon Sep 17 00:00:00 2001 From: Alexandr-Solovev Date: Mon, 5 Oct 2026 04:01:07 -0700 Subject: [PATCH 6/9] configs(hdbscan): add the HDBSCAN vs DBSCAN comparison experiment Three templates over the same four datasets and the same data format: DBSCAN at the 1% distance quantile, HDBSCAN, and HDBSCAN re-cut at that same distance through `dbscan_clustering`. The third one answers what the extra cost of the hierarchy buys - one fit serves every distance, where DBSCAN has to run again. `ensure_sklearnex_patching` is off for the re-cut template because sklearnex does not patch `dbscan_clustering`; it runs the scikit-learn method over the single linkage tree that the patched `fit` produced. Co-Authored-By: Claude Opus 5 --- configs/experiments/hdbscan_vs_dbscan.json | 89 ++++++++++++++++++++++ 1 file changed, 89 insertions(+) create mode 100644 configs/experiments/hdbscan_vs_dbscan.json diff --git a/configs/experiments/hdbscan_vs_dbscan.json b/configs/experiments/hdbscan_vs_dbscan.json new file mode 100644 index 00000000..09452fa6 --- /dev/null +++ b/configs/experiments/hdbscan_vs_dbscan.json @@ -0,0 +1,89 @@ +{ + "INCLUDE": ["../common/sklearn.json", "../common/hdbscan.json", "../common/dbscan.json"], + "PARAMETERS_SETS": { + "hdbscan re-cut as dbscan": { + "algorithm": { + "estimator_methods": { "training": "fit|dbscan_clustering" }, + "method_params": { + "dbscan_clustering": { + "cut_distance": "[SPECIAL_VALUE]distances_quantile:0.01", + "min_cluster_size": 5 + } + } + }, + "bench": { "ensure_sklearnex_patching": false } + }, + "density clustering comparison datasets": { + "data": [ + { + "dataset": "skin_segmentation", + "split_kwargs": { "train_size": 100000 }, + "preprocessing_kwargs": { "normalize": "standard" } + }, + { + "dataset": "mnist", + "split_kwargs": { "train_size": 20000 }, + "preprocessing_kwargs": { "normalize": "standard" } + }, + { + "source": "make_blobs", + "generation_kwargs": { + "centers": 10, + "n_samples": 50000, + "n_features": 8, + "cluster_std": 2.0, + "random_state": 42 + }, + "split_kwargs": { "ignore": true } + }, + { + "source": "make_blobs", + "generation_kwargs": { + "centers": 10, + "n_samples": 25000, + "n_features": 64, + "cluster_std": 4.0, + "random_state": 42 + }, + "split_kwargs": { "ignore": true } + } + ] + }, + "density clustering comparison data format": { + "data": { "format": "numpy", "order": "C", "dtype": "float64" }, + "bench": { "n_runs": 3, "time_limit": 1200 } + } + }, + "TEMPLATES": { + "dbscan": { + "SETS": [ + "common dbscan parameters", + "sklearn dbscan parameters", + "density clustering comparison datasets", + "density clustering comparison data format", + "sklearn-ex[cpu,gpu] implementations" + ] + }, + "hdbscan": { + "SETS": [ + "common hdbscan parameters", + "sklearn hdbscan parameters", + "hdbscan kd_tree method", + "density clustering comparison datasets", + "density clustering comparison data format", + "sklearn-ex[cpu,gpu] implementations" + ] + }, + "hdbscan re-cut at the dbscan distance": { + "SETS": [ + "common hdbscan parameters", + "sklearn hdbscan parameters", + "hdbscan kd_tree method", + "hdbscan re-cut as dbscan", + "density clustering comparison datasets", + "density clustering comparison data format", + "sklearn-ex[cpu] implementations" + ] + } + } +} From 7ef6e8c5d15ca131e6410d54100ababa71b769d6 Mon Sep 17 00:00:00 2001 From: Alexandr-Solovev Date: Mon, 5 Oct 2026 04:01:07 -0700 Subject: [PATCH 7/9] fix: take DBSCAN cluster metrics from the host copy of the labels `np.unique(y)` on the raw `y` raises for a `dpnp` array, so every DBSCAN case on a GPU implementation set failed before reporting homogeneity and completeness. The host copy `y_compat` is already at hand for exactly this. Co-Authored-By: Claude Opus 5 --- sklbench/benchmarks/sklearn_estimator.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/sklbench/benchmarks/sklearn_estimator.py b/sklbench/benchmarks/sklearn_estimator.py index 99fbda1e..9fc70094 100644 --- a/sklbench/benchmarks/sklearn_estimator.py +++ b/sklbench/benchmarks/sklearn_estimator.py @@ -225,7 +225,7 @@ def get_subset_metrics_of_estimator( ) } ) - if len(np.unique(y)) < 128: + if len(np.unique(y_compat)) < 128: metrics.update( { "homogeneity": ( From e06b66fbf4c5777e803da421c96a3cf4504def0f Mon Sep 17 00:00:00 2001 From: Alexandr-Solovev Date: Mon, 5 Oct 2026 04:39:31 -0700 Subject: [PATCH 8/9] configs(hdbscan): split the comparison datasets by neighbors method A kd_tree over MNIST's 784 columns is the wrong method for the dataset and costs scikit-learn two and a half minutes per run. Each of the three comparisons now covers a tree-friendly pair (skin_segmentation, blobs 50k x 8) with kd_tree and a brute-friendly pair (MNIST, blobs 25k x 64) with brute force, so HDBSCAN runs at the method the dimension calls for. Co-Authored-By: Claude Opus 5 --- configs/experiments/hdbscan_vs_dbscan.json | 66 ++++++++++++++++------ 1 file changed, 50 insertions(+), 16 deletions(-) diff --git a/configs/experiments/hdbscan_vs_dbscan.json b/configs/experiments/hdbscan_vs_dbscan.json index 09452fa6..a8a876d3 100644 --- a/configs/experiments/hdbscan_vs_dbscan.json +++ b/configs/experiments/hdbscan_vs_dbscan.json @@ -13,18 +13,13 @@ }, "bench": { "ensure_sklearnex_patching": false } }, - "density clustering comparison datasets": { + "density comparison datasets [tree]": { "data": [ { "dataset": "skin_segmentation", "split_kwargs": { "train_size": 100000 }, "preprocessing_kwargs": { "normalize": "standard" } }, - { - "dataset": "mnist", - "split_kwargs": { "train_size": 20000 }, - "preprocessing_kwargs": { "normalize": "standard" } - }, { "source": "make_blobs", "generation_kwargs": { @@ -35,6 +30,15 @@ "random_state": 42 }, "split_kwargs": { "ignore": true } + } + ] + }, + "density comparison datasets [brute]": { + "data": [ + { + "dataset": "mnist", + "split_kwargs": { "train_size": 20000 }, + "preprocessing_kwargs": { "normalize": "standard" } }, { "source": "make_blobs", @@ -49,39 +53,69 @@ } ] }, - "density clustering comparison data format": { + "density comparison data format": { "data": { "format": "numpy", "order": "C", "dtype": "float64" }, "bench": { "n_runs": 3, "time_limit": 1200 } } }, "TEMPLATES": { - "dbscan": { + "dbscan [tree-friendly data]": { + "SETS": [ + "common dbscan parameters", + "sklearn dbscan parameters", + "density comparison datasets [tree]", + "density comparison data format", + "sklearn-ex[cpu,gpu] implementations" + ] + }, + "dbscan [brute-friendly data]": { "SETS": [ "common dbscan parameters", "sklearn dbscan parameters", - "density clustering comparison datasets", - "density clustering comparison data format", + "density comparison datasets [brute]", + "density comparison data format", "sklearn-ex[cpu,gpu] implementations" ] }, - "hdbscan": { + "hdbscan [tree-friendly data]": { "SETS": [ "common hdbscan parameters", "sklearn hdbscan parameters", "hdbscan kd_tree method", - "density clustering comparison datasets", - "density clustering comparison data format", + "density comparison datasets [tree]", + "density comparison data format", + "sklearn-ex[cpu,gpu] implementations" + ] + }, + "hdbscan [brute-friendly data]": { + "SETS": [ + "common hdbscan parameters", + "sklearn hdbscan parameters", + "hdbscan brute method", + "density comparison datasets [brute]", + "density comparison data format", "sklearn-ex[cpu,gpu] implementations" ] }, - "hdbscan re-cut at the dbscan distance": { + "hdbscan re-cut at the dbscan distance [tree-friendly data]": { "SETS": [ "common hdbscan parameters", "sklearn hdbscan parameters", "hdbscan kd_tree method", "hdbscan re-cut as dbscan", - "density clustering comparison datasets", - "density clustering comparison data format", + "density comparison datasets [tree]", + "density comparison data format", + "sklearn-ex[cpu] implementations" + ] + }, + "hdbscan re-cut at the dbscan distance [brute-friendly data]": { + "SETS": [ + "common hdbscan parameters", + "sklearn hdbscan parameters", + "hdbscan brute method", + "hdbscan re-cut as dbscan", + "density comparison datasets [brute]", + "density comparison data format", "sklearn-ex[cpu] implementations" ] } From 9ad201f985d6b70c3bfcccff624f26670a05b3ab Mon Sep 17 00:00:00 2001 From: Alexandr-Solovev Date: Mon, 5 Oct 2026 04:59:49 -0700 Subject: [PATCH 9/9] fix: score the labels a data-free estimator method returns 'HDBSCAN.dbscan_clustering' re-cuts the fitted hierarchy and returns the new labels without touching 'labels_', so the clustering metrics reported for it were a copy of the ones from 'fit'. Co-Authored-By: Claude Opus 5 --- sklbench/benchmarks/sklearn_estimator.py | 67 +++++++++++++++--------- 1 file changed, 42 insertions(+), 25 deletions(-) diff --git a/sklbench/benchmarks/sklearn_estimator.py b/sklbench/benchmarks/sklearn_estimator.py index 9fc70094..505e2f02 100644 --- a/sklbench/benchmarks/sklearn_estimator.py +++ b/sklbench/benchmarks/sklearn_estimator.py @@ -114,6 +114,31 @@ def get_number_of_classes(estimator_instance, y): return len(np.unique(y)) +def get_clustering_metrics_of_labels(labels, x_compat, y_compat) -> Dict[str, float]: + """Score a cluster assignment produced by a clustering estimator. + + @param labels Cluster index per sample, `-1` for noise + @param x_compat Clustered data as a numpy array + @param y_compat Ground truth labels as a numpy array + + @return Cluster count and, for more than one cluster, the Davies-Bouldin, + homogeneity and completeness scores + """ + labels = convert_to_numpy(labels) + clusters = len(np.unique(labels[labels != -1])) + metrics = {"clusters": clusters} + if clusters > 1: + metrics["Davies-Bouldin score"] = float(davies_bouldin_score(x_compat, labels)) + if len(np.unique(y_compat)) < 128: + metrics["homogeneity"] = ( + float(homogeneity_score(y_compat, labels)) if clusters > 1 else 0 + ) + metrics["completeness"] = ( + float(completeness_score(y_compat, labels)) if clusters > 1 else 0 + ) + return metrics + + def get_subset_metrics_of_estimator( task, stage, estimator_instance, data ) -> Dict[str, float]: @@ -214,32 +239,11 @@ def get_subset_metrics_of_estimator( } ) if "DBSCAN" in str(estimator_instance) and stage == "training": - labels = convert_to_numpy(estimator_instance.labels_) - clusters = len(np.unique(labels[labels != -1])) - metrics.update({"clusters": clusters}) - if clusters > 1: - metrics.update( - { - "Davies-Bouldin score": float( - davies_bouldin_score(x_compat, labels) - ) - } - ) - if len(np.unique(y_compat)) < 128: - metrics.update( - { - "homogeneity": ( - float(homogeneity_score(y_compat, labels)) - if clusters > 1 - else 0 - ), - "completeness": ( - float(completeness_score(y_compat, labels)) - if clusters > 1 - else 0 - ), - } + metrics.update( + get_clustering_metrics_of_labels( + estimator_instance.labels_, x_compat, y_compat ) + ) elif task == "manifold": if hasattr(estimator_instance, "kl_divergence_") and stage == "training": metrics.update( @@ -444,6 +448,9 @@ def measure_sklearn_estimator( sklearnex_logging_stream = get_sklearnex_logging_stream() metrics = dict() + # quality metrics of the labels a data-free method returns itself, which the + # per-stage metrics below cannot see + method_label_metrics = dict() estimator_instance = estimator_class(**estimator_params) for stage in estimator_methods.keys(): for method in estimator_methods[stage]: @@ -495,6 +502,14 @@ def measure_sklearn_estimator( metrics[method] = measure_case( bench_case, method_instance, *data_args, **(method_params or {}) ) + if method_params is not None and task == "clustering": + # `labels_` still holds the 'fit' partition, so the returned + # labels are the only description of what this method did + method_label_metrics[method] = get_clustering_metrics_of_labels( + getattr(estimator_instance, method)(**method_params), + convert_to_numpy(x_train), + convert_to_numpy(y_train), + ) if ensure_sklearnex_patching: full_method_name = f"{estimator_class.__name__}.{method}" sklearnex_logging_stream.seek(0) @@ -518,6 +533,8 @@ def measure_sklearn_estimator( for stage in estimator_methods.keys(): if method in estimator_methods[stage]: metrics[method].update(quality_metrics[stage]) + if method in method_label_metrics: + metrics[method].update(method_label_metrics[method]) return metrics, estimator_instance