Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions configs/BENCH-CONFIG-SPEC.md
Original file line number Diff line number Diff line change
Expand Up @@ -108,6 +108,7 @@ Configs have the three highest parameter keys:
|:---------------|:--------------|:--------|:------------|
| `algorithm`:`estimator` | None | | Name of measured estimator. |
| `algorithm`:`estimator_params` | Empty `dict` | | Parameters for estimator constructor. |
| `algorithm`:`method_params`:`{method}` | None | | Parameters for a measured method of estimator. A method listed here is called with these parameters only, without data, which fits methods working off the fitted model such as `HDBSCAN.dbscan_clustering`. |
| `algorithm`:`batch_size`:`{stage}` | None | Any positive integer | Enables online mode for `{stage}` methods of estimator (sequential calls for each batch). |
| `algorithm`:`sklearn_context` | None | | Parameters for sklearn `config_context` used over estimator. `array_api_dispatch` requires `SCIPY_ARRAY_API=1`, which scikit-learn_bench sets by default if it is unset in the environment. |
| `algorithm`:`sklearnex_context` | None | | Parameters for sklearnex `config_context` used over estimator. Updated by `sklearn_context` if set. |
Expand Down Expand Up @@ -140,6 +141,7 @@ List of available special values:
| `algorithm`:`estimator_params`:`scale_pos_weight` | sklearn_estimator | `auto` | Sets `scale_pos_weight` parameter to `sum(negative instances) / sum(positive instances)` value for estimator. |
| `algorithm`:`estimator_params`:`n_clusters` | sklearn_estimator | `auto` | Sets `n_clusters` parameter to number of clusters or classes from dataset description for estimator. |
| `algorithm`:`estimator_params`:`eps` | sklearn_estimator | `distances_quantile:{quantile}` format where quantile is *float* value in [0, 1] range | Computes `eps` parameter as quantile value of distances in `x_train` matrix for estimator. |
| `algorithm`:`method_params`:`dbscan_clustering`:`cut_distance` | sklearn_estimator | `distances_quantile:{quantile}` format where quantile is *float* value in [0, 1] range | Computes `cut_distance` the same way as `eps` above, so that an HDBSCAN hierarchy can be re-cut at the distance DBSCAN is measured at. |

## Range of Values

Expand Down
48 changes: 48 additions & 0 deletions configs/common/hdbscan.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,48 @@
{
"PARAMETERS_SETS": {
"common hdbscan parameters": {
"algorithm": {
"estimator": "HDBSCAN",
"estimator_params": {
"min_cluster_size": 5,
"min_samples": 5,
"metric": "euclidean",
"cluster_selection_method": "eom",
"allow_single_cluster": false,
"store_centers": null,
"copy": false
},
"estimator_methods": { "training": "fit" }
},
"data": { "format": "numpy", "order": "C", "dtype": "float64" },
"bench": { "n_runs": 3, "time_limit": 1200 }
},
"sklearn hdbscan parameters": {
"algorithm": {
"estimator_params": { "n_jobs": "[SPECIAL_VALUE]physical_cpus" }
}
},
"hdbscan brute method": {
"algorithm": { "estimator_params": { "algorithm": "brute" } }
},
"hdbscan kd_tree method": {
"algorithm": { "estimator_params": { "algorithm": "kd_tree", "leaf_size": 40 } }
},
"hdbscan ball_tree method": {
"algorithm": { "estimator_params": { "algorithm": "ball_tree", "leaf_size": 40 } }
},
"hdbscan tree methods": {
"algorithm": {
"estimator_params": { "algorithm": ["kd_tree", "ball_tree"], "leaf_size": 40 }
}
},
"hdbscan all methods": {
"algorithm": {
"estimator_params": {
"algorithm": ["brute", "kd_tree", "ball_tree"],
"leaf_size": 40
}
}
}
}
}
4 changes: 4 additions & 0 deletions configs/experiments/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -2,4 +2,8 @@

`daal4py_svd`: tests performance scalability of `daal4py.svd` algorithm

`hdbscan_parameters`: sweeps the `HDBSCAN` parameter space (metrics, cluster selection, density thresholds, stored centers, dtypes and data formats) over the `sklearn` and `sklearnex` implementations.

`hdbscan_scaling`: tests thread, NUMA, `n_samples` and `n_features` scalability of `HDBSCAN`.

`nearest_neighbors`: tests performance of neighbors search implementations from `sklearnex`, `sklearn`, `raft`, `faiss` and `svs`.
169 changes: 169 additions & 0 deletions configs/experiments/hdbscan_parameters.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,169 @@
{
"INCLUDE": ["../common/sklearn.json", "../common/hdbscan.json"],
"PARAMETERS_SETS": {
"hdbscan tree metrics": {
"algorithm": {
"estimator_params": { "metric": ["euclidean", "manhattan", "chebyshev"] }
}
},
"hdbscan minkowski metric": {
"algorithm": {
"estimator_params": { "metric": "minkowski", "metric_params": { "p": 3 } }
}
},
"hdbscan cosine metric": {
"algorithm": {
"estimator_params": { "metric": "cosine", "algorithm": "brute" }
}
},
"hdbscan cluster selection sweep": {
"algorithm": {
"estimator_params": { "cluster_selection_method": ["eom", "leaf"] }
}
},
"hdbscan density sweep": {
"algorithm": {
"estimator_params": { "min_cluster_size": [5, 25, 100], "min_samples": [5, 25] }
}
},
"hdbscan store centers sweep": {
"algorithm": {
"estimator_params": { "store_centers": [null, "centroid", "medoid", "both"] }
}
},
"hdbscan dtype sweep": {
"data": { "dtype": ["float32", "float64"] }
},
"hdbscan data format sweep": {
"data": [
{ "format": "numpy", "order": "C" },
{ "format": "numpy", "order": "F" },
{ "format": "pandas", "order": "F" }
]
},
"hdbscan parameters data [tree]": {
"data": {
"source": "make_blobs",
"generation_kwargs": {
"centers": 10,
"n_samples": 100000,
"n_features": 8,
"cluster_std": 2.0,
"random_state": 42
},
"split_kwargs": { "ignore": true }
}
},
"hdbscan parameters data [brute]": {
"data": {
"source": "make_blobs",
"generation_kwargs": {
"centers": 10,
"n_samples": 25000,
"n_features": 64,
"cluster_std": 4.0,
"random_state": 42
},
"split_kwargs": { "ignore": true }
}
},
"hdbscan parameters data [real]": {
"data": {
"dataset": "skin_segmentation",
"split_kwargs": { "train_size": 30000 },
"preprocessing_kwargs": { "normalize": "standard" }
}
}
},
"TEMPLATES": {
"hdbscan metrics [tree]": {
"SETS": [
"sklearn-ex[cpu] implementations",
"common hdbscan parameters",
"sklearn hdbscan parameters",
"hdbscan tree methods",
"hdbscan tree metrics",
"hdbscan parameters data [tree]"
]
},
"hdbscan minkowski [tree]": {
"SETS": [
"sklearn-ex[cpu] implementations",
"common hdbscan parameters",
"sklearn hdbscan parameters",
"hdbscan kd_tree method",
"hdbscan minkowski metric",
"hdbscan parameters data [tree]"
]
},
"hdbscan metrics [brute]": {
"SETS": [
"sklearn-ex[cpu] implementations",
"common hdbscan parameters",
"sklearn hdbscan parameters",
"hdbscan brute method",
"hdbscan tree metrics",
"hdbscan parameters data [brute]"
]
},
"hdbscan cosine [brute]": {
"SETS": [
"sklearn-ex[cpu] implementations",
"common hdbscan parameters",
"sklearn hdbscan parameters",
"hdbscan cosine metric",
"hdbscan parameters data [brute]"
]
},
"hdbscan cluster selection": {
"SETS": [
"sklearn-ex[cpu] implementations",
"common hdbscan parameters",
"sklearn hdbscan parameters",
"hdbscan kd_tree method",
"hdbscan cluster selection sweep",
"hdbscan parameters data [tree]"
]
},
"hdbscan density": {
"SETS": [
"sklearn-ex[cpu] implementations",
"common hdbscan parameters",
"sklearn hdbscan parameters",
"hdbscan kd_tree method",
"hdbscan density sweep",
"hdbscan parameters data [tree]"
]
},
"hdbscan store centers": {
"SETS": [
"sklearn-ex[cpu] implementations",
"common hdbscan parameters",
"sklearn hdbscan parameters",
"hdbscan kd_tree method",
"hdbscan store centers sweep",
"hdbscan parameters data [tree]"
]
},
"hdbscan dtypes": {
"SETS": [
"sklearn-ex[cpu] implementations",
"common hdbscan parameters",
"sklearn hdbscan parameters",
"hdbscan all methods",
"hdbscan dtype sweep",
"hdbscan parameters data [real]"
]
},
"hdbscan data formats": {
"SETS": [
"sklearn-ex[cpu] implementations",
"common hdbscan parameters",
"sklearn hdbscan parameters",
"hdbscan kd_tree method",
"hdbscan data format sweep",
"hdbscan parameters data [tree]"
]
}
}
}
Loading
Loading