"""RAPIDS cuML where it helps, and the CPU implementation everywhere else.
Install the optional RAPIDS support with:
pip install spacr[rapids]
and nothing else changes. The CPU path stays the default, stays tested, and
stays the answer whenever cuML is absent, the interpreter is wrong, there is
no CUDA device, or the caller asks for determinism.
WHY AN EXTRA AND NEVER A DEPENDENCY. ``cuml-cu12`` declares
``requires_python >= 3.11`` with classifiers for 3.11 and 3.12 ONLY, and
wants ``numpy>=2.0`` and ``scipy>=1.14``. spaCR promises 3.9 through 3.14, so
making it core would drop four of six interpreters. As an extra it constrains
nothing -- see the note beside it in setup.py.
WHERE IT ACTUALLY HELPS. cuML implements the algorithms spaCR already runs on
big tables: UMAP, t-SNE, PCA, DBSCAN and KMeans. Those are the ones offered
here. Everything else spaCR does -- barcode mapping, format conversion,
SQLite, report assembly, grouped statistics -- is decompression, filesystem
and small-table work, and moving it to a GPU would cost transfer time and buy
nothing. GPU acceleration is reserved for workloads where transfer overhead
is outweighed by computation.
**Determinism is a real difference, not a footnote.** cuML's UMAP is not
bit-identical to umap-learn's, and its KMeans and DBSCAN can differ at the
boundaries. A figure regenerated on a different machine would move. So the
accelerator is OPT-IN per call, reports which backend ran, and any caller
that pins a seed for reproducibility should keep the CPU path.
"""
from __future__ import annotations
import logging
import os
from typing import Any, Dict, List, Optional, Tuple
LOG = logging.getLogger("spacr.gpu_reduce")
#: The algorithms cuML is offered for. Each is one spaCR already runs and one
#: whose cuML implementation takes the same shaped input.
ACCELERATED: Tuple[str, ...] = ("umap", "tsne", "pca", "dbscan", "kmeans")
#: Set to anything falsy to force the CPU path regardless of what is
#: installed. The escape hatch for "the GPU answer looks wrong" that does not
#: require uninstalling anything.
ENV_FLAG = "SPACR_USE_RAPIDS"
[docs]
def rapids_available() -> bool:
"""Is cuML importable AND is there a device for it?
Both halves matter: cuML imports happily on a machine with no GPU and
then fails at fit time, which would turn an optional accelerator into a
crash on exactly the machines that did not ask for one.
"""
if not _env_allows():
return False
try:
import cuml # noqa: F401
except Exception:
return False
try:
import cupy
return bool(cupy.cuda.runtime.getDeviceCount())
except Exception:
return False
def _env_allows() -> bool:
"""Return whether :data:`ENV_FLAG` is unset or not a recognized false value."""
raw = os.environ.get(ENV_FLAG)
if raw is None:
return True
return str(raw).strip().lower() not in ("0", "false", "no", "off", "")
[docs]
def backend_for(method: str, *, prefer_gpu: bool = False) -> str:
"""``'cuml'`` or ``'cpu'`` for ``method``, and never a surprise.
:param method: reducer name. Only algorithms in :data:`ACCELERATED` can
select cuML; unsupported names remain on the CPU path.
:param prefer_gpu: opt in. Default False, so an existing caller keeps the
CPU path and the reproducibility that goes with it.
:returns: the backend that will actually run.
"""
if not prefer_gpu:
return "cpu"
if str(method).strip().lower() not in ACCELERATED:
return "cpu"
return "cuml" if rapids_available() else "cpu"
[docs]
def make_reducer(method: str, *, prefer_gpu: bool = False, **kwargs) -> Tuple[Any, str]:
"""Build the estimator for ``method``, on whichever backend is available.
:param method: reducer name understood by the cuML or CPU estimator
factories (``umap``, ``tsne``, ``pca``, ``dbscan`` or ``kmeans``).
:param kwargs: passed to the estimator. The parameter names cuML shares
with the CPU libraries -- ``n_neighbors``, ``min_dist``,
``n_components``, ``eps``, ``min_samples``, ``n_clusters`` -- carry
through unchanged, which is what makes one call site serve both.
:returns: ``(estimator, backend)``. The backend is returned rather than
logged only, so a caller can record WHICH one produced a figure.
:raises ImportError: the CPU library for ``method`` is missing. A missing
optional GPU is a fallback; a missing required CPU library is a
genuine setup problem and is not silently worked around.
"""
name = str(method).strip().lower()
backend = backend_for(name, prefer_gpu=prefer_gpu)
if backend == "cuml":
try:
return _cuml_estimator(name, **kwargs), "cuml"
except Exception:
LOG.info("cuML could not build a %s estimator; using the CPU "
"implementation", name, exc_info=True)
return _cpu_estimator(name, **kwargs), "cpu"
def _cuml_estimator(name: str, **kwargs):
"""Construct the named cuML reducer or clusterer with ``kwargs``."""
import cuml
if name == "umap":
return cuml.UMAP(**kwargs)
if name == "tsne":
return cuml.TSNE(**kwargs)
if name == "pca":
return cuml.PCA(**kwargs)
if name == "dbscan":
return cuml.DBSCAN(**kwargs)
if name == "kmeans":
return cuml.KMeans(**kwargs)
raise ValueError(f"{name!r} has no cuML equivalent here")
def _cpu_estimator(name: str, **kwargs):
"""Construct the named CPU reducer or clusterer with ``kwargs``."""
if name == "umap":
from .utils import umap
return umap.UMAP(**kwargs)
if name == "tsne":
from sklearn.manifold import TSNE
return TSNE(**kwargs)
if name == "pca":
from sklearn.decomposition import PCA
return PCA(**kwargs)
if name == "dbscan":
from sklearn.cluster import DBSCAN
return DBSCAN(**kwargs)
if name == "kmeans":
from sklearn.cluster import KMeans
return KMeans(**kwargs)
raise ValueError(f"{name!r} is not one of {list(ACCELERATED)}")
#: The interpreters ``cuml-cu12`` declares. Not a guess -- read off the wheel
#: metadata, which carries classifiers for 3.11 and 3.12 ONLY. On anything
#: else pip produces a resolver error a user cannot act on, so spaCR says
#: what is needed instead of letting pip say what went wrong.
SUPPORTED_PYTHON = ((3, 11), (3, 12))
[docs]
def python_supported() -> bool:
"""Can cuML be installed into the interpreter running this?"""
import sys as _sys
return _sys.version_info[:2] in SUPPORTED_PYTHON
[docs]
def install_plan() -> Dict[str, Any]:
"""What pressing GPU should do, decided before anything is installed.
:returns: ``{action, message}``. ``action`` is ``ready`` when cuML and a
device are available; ``install`` when this interpreter can install
it; ``wrong_python`` when another Python version is required; or
``no_device`` when installing more cannot provide a CUDA device.
NOTHING IS INSTALLED HERE. This function decides and reports; the caller
installs, because installing is the part that needs a confirmation and a
progress bar, and a function that did both could not be asked "what would
happen" without it happening.
"""
import sys as _sys
version = f"{_sys.version_info.major}.{_sys.version_info.minor}"
if rapids_available():
return {"action": "ready", "message": describe()}
try:
import cuml # noqa: F401
return {"action": "no_device",
"message": ("cuML is installed but no CUDA device answered. "
"Check the driver with nvidia-smi -- installing "
"again cannot fix a missing device.")}
except Exception:
pass
if not python_supported():
wanted = " or ".join(f"{a}.{b}" for a, b in SUPPORTED_PYTHON)
return {"action": "wrong_python",
"message": (f"cuML supports Python {wanted} only, and this is "
f"{version}. Make a {SUPPORTED_PYTHON[0][0]}."
f"{SUPPORTED_PYTHON[0][1]} environment and "
f"install spaCR there:\n\n"
f" conda create -n spacr-gpu python="
f"{SUPPORTED_PYTHON[0][0]}."
f"{SUPPORTED_PYTHON[0][1]}\n"
f" conda activate spacr-gpu\n"
f" pip install 'spacr[rapids]'")}
return {"action": "install",
"message": ("Install cuML for GPU UMAP?\n\nThis downloads "
"SEVERAL GIGABYTES of CUDA libraries -- cuml-cu12 "
"pulls libcuml, cudf, cupy and the CUDA runtime. It "
"is not a small wheel and it will take a while.\n\n"
"spaCR must be RESTARTED afterwards: pip can upgrade "
"numpy and scipy underneath a process that has "
"already imported them, and this one has.")}
[docs]
def install_command() -> List[str]:
"""The command that installs the extra. Separate so it can be shown."""
import sys as _sys
return [_sys.executable, "-m", "pip", "install", "spacr[rapids]"]
[docs]
def describe() -> str:
"""One line for a log or an About box: what is available, and why not."""
if not _env_allows():
return f"RAPIDS disabled by {ENV_FLAG}"
try:
import cuml
except Exception:
return ("RAPIDS not installed (pip install 'spacr[rapids]', "
"Python 3.11 or 3.12)")
try:
import cupy
devices = cupy.cuda.runtime.getDeviceCount()
except Exception:
devices = 0
if not devices:
return f"cuML {getattr(cuml, '__version__', '?')} installed, no CUDA device"
return f"cuML {getattr(cuml, '__version__', '?')} on {devices} device(s)"
#: The requirement that actually installs the accelerator. ``install_command``
#: above asks for ``spacr[rapids]`` because that is the documented spelling of
#: the extra; the resolver is given the concrete wheel name, so a dry-run
#: report names the package a user can look up.
RAPIDS_REQUIREMENT = "cuml-cu12"
[docs]
def install_offer():
"""The same offer :func:`install_plan` describes, in the shared shape.
The Image UMAP's GPU acceleration and the regression backend picker ask
the same question, so they answer it in the same
vocabulary and one hover panel serves both. This is the bridge --
:func:`install_plan` keeps its own dict because the Hyperparameter screen
already reads it.
:returns: a :class:`spacr.updater.InstallOffer`, whose ``action`` is
``ready``, ``install``, ``elsewhere`` or ``impossible``.
"""
from .regression_backends import INSTALL_RECIPES
from .updater import (offer_elsewhere, offer_impossible, offer_install,
offer_ready)
recipe = INSTALL_RECIPES.get('cuml', "")
plan = install_plan()
action, message = plan["action"], plan["message"]
title = "GPU acceleration (cuML)"
if action == "ready":
return offer_ready(title, message)
if action == "no_device":
return offer_impossible(title, message, recipe)
if action == "wrong_python":
return offer_elsewhere(title, message, recipe)
return offer_install(title, message, RAPIDS_REQUIREMENT, recipe)
[docs]
def availability_entry() -> Dict[str, Any]:
"""GPU acceleration as the shared hover panel wants it.
Mirrors :func:`spacr.regression_backends.availability_entry`, so the panel
takes one mapping shape and neither caller imports the other.
:returns: ``{key, title, reason, url, offer, enabled}``.
"""
offer = install_offer()
return {
'key': 'cuml',
'title': "GPU acceleration (cuML)",
'reason': offer.message,
'url': "https://docs.rapids.ai/api/cuml/stable/",
'enabled': offer.action == "ready",
'offer': offer,
}