Source code for spacr.classify

"""One Classify entry point over both classifier families.

**What it is for.** Classify is the fourth step of the pipeline, after Mask,
Measure and Annotate. It learns the classes a screen is scored on and
predicts them for every object. ``classifier_family`` chooses how: ``cv``
trains an image model on object crops through
:func:`spacr.deep_spacr.deep_spacr`, and ``ml`` fits a classical model such as
XGBoost, LightGBM or a random forest on measured features through
:func:`spacr.ml.generate_ml_scores`.

**What it needs.** A project that has been through Mask and Measure, so that
``measurements/measurements.db`` and its object crops exist, and a definition
of the classes. ``dataset_mode`` sets that definition for both families:
``metadata`` takes classes from plate metadata such as the positive and
negative control wells, and ``annotation`` takes them from an annotation
column of ``png_list``, as written by Annotate. The CV family reads crops from
PNG files or cuts them from the ``merged`` arrays, as ``crop_source`` selects.

**What it produces.** A CV run writes model checkpoints,
``DL_model_settings.csv``, a dataset tar, a ``top_examples/`` folder with the
most confident crops of each class and an evaluation bundle with the held-out
performance and ``leakage.json``, and merges each object's predicted class and
probability into ``png_list``. An ML run writes per-object predictions,
feature-importance and permutation tables and a plate heatmap to
``results/`` beside the measurements database.

**What to do next.** Check the held-out performance, and the leakage verdict
on the QC screen, before trusting the predictions. Then take per-well scores
into Regression, which pairs them with the guide counts from Map Barcodes to
estimate which guides and genes explain the phenotype.

Classify (CV) trains a Torch model on object crops. Classify (ML) fits a
gradient-boosted model on measured features. They answer the same question --
*which class is this object* -- and until now they were two modules that
shared six setting names out of 78 and 37, two of which disagreed on their
default.

:mod:`spacr.training_basis` already unified what defines a CLASS. This module
unifies what runs: one settings dict, one ``classifier_family`` switch, one
call. The two original modules stay exactly as they are -- a merged screen
that removed them would strand every saved settings CSV and every notebook
that imports their entry points.

**Nothing here reimplements either pipeline.** ``deep_spacr`` and
``generate_ml_scores`` are called unchanged, which is what makes the merged
module honest: a run through it and a run through the module it dispatches to
produce the same result, because they are the same code.
"""
from __future__ import annotations

import os
import sys
from typing import Any, Dict, Mapping, Tuple

#: The two classifier families, in the order the settings panel offers them.
CLASSIFIER_FAMILIES: Tuple[str, ...] = ("cv", "ml")

ML_MODEL_TYPES: Tuple[str, ...] = (
    "xgboost", "lightgbm", "catboost", "random_forest", "extra_trees",
    "gradient_boosting", "logistic_regression", "svm", "mlp",
)

#: family -> the app key whose settings and pipeline it uses. The merged
#: screen is a front end onto these, not a replacement for them.
FAMILY_APP_KEY: Dict[str, str] = {"cv": "classify", "ml": "ml_analyze"}

#: Settings each family reads that the other has no use for. Drives the
#: greying, the same way :data:`spacr.training_basis.BASIS_SETTINGS` does for
#: the training basis, and for the same reason: a control the user can edit
#: that changes nothing is worse than one that is not there.
FAMILY_SETTINGS: Dict[str, Tuple[str, ...]] = {
    "cv": (
        "crop_shape", "extract_channels", "object_array", "coordinate_columns",
        "model_type", "custom_model_path", "image_size",
        "train_channels", "epochs", "optimizer_type", "schedule", "loss_type",
        "dropout_rate", "init_weights", "amsgrad", "weight_decay",
        "gradient_accumulation_steps",
        "early_stopping_patience", "augment", "pin_memory", "use_checkpoint",
        "resume_checkpoint", "tensorboard", "focal_gamma", "focal_alpha",
        "label_smoothing", "logit_adjust_tau", "train", "test",
        "generate_training_dataset", "apply_model_to_dataset",
        "generate_full_dataset", "tar_path", "n_top_examples", "path_string",
        "file_type", "crop_source", "batch_size", "val_split",
        "image_source", "mixed_precision",
    ),
    "ml": (
        "model_type_ml", "n_estimators", "reg_alpha", "reg_lambda",
        "prune_features", "top_features", "n_repeats", "min_cells_per_well",
        "remove_low_variance_features", "remove_highly_correlated_features",
        "heatmap_feature", "grouping", "min_max", "cmap",
        "batch_correction", "batch_column", "batch_control_column",
        "batch_control_values", "batch_covariate_column",
        "batch_combat_mean_only", "batch_min_samples", "batch_missing_control",
        "nuclei_limit", "pathogen_limit", "exclude",
    ),
}


[docs] class ClassifierFamilyError(ValueError): """A classifier family spaCR does not have."""
def _begin_flowview_run(settings: Mapping[str, Any]) -> object | None: """Start a live graph only when optional FlowView tracing is enabled. :param settings: Raw Classify settings fingerprinted into the new FlowView graph. :returns: The newly installed collector when tracing is enabled, otherwise ``None``. The common disabled path is a module-cache lookup and an environment check; importantly, it imports no FlowView code. A panel can enable the already-loaded trace module, while ``SPACR_FLOWVIEW`` opts a headless run in through the same lazy boundary. """ trace_module = sys.modules.get("spacr.flowview.trace") if trace_module is None: enabled_by_environment = os.environ.get("SPACR_FLOWVIEW", "") if enabled_by_environment.strip().casefold() not in { "1", "on", "true", "yes", }: return None from .flowview import trace as trace_module if not trace_module.is_enabled(): return None from .flowview.classify_blueprint import _install_classify_collector return _install_classify_collector(settings)
[docs] def resolve_family(settings: Mapping[str, Any]) -> str: """Return the classifier family a settings dict asks for. Defaults to ``'cv'``, because the merged module's own default settings are the CV ones and a dict with no family is most likely a Classify (CV) CSV opened in the merged screen. :param settings: the run settings. :returns: ``'cv'`` or ``'ml'``. :raises ClassifierFamilyError: an unrecognised family. Guessing would train a different kind of model than the user asked for and report success. """ declared = settings.get("classifier_family") if declared is None or declared == "": return "cv" family = str(declared).strip().lower() if family not in CLASSIFIER_FAMILIES: raise ClassifierFamilyError( f"classifier_family={declared!r} is not one of " f"{list(CLASSIFIER_FAMILIES)}") return family
[docs] def inapplicable_settings(family: str) -> Tuple[str, ...]: """Settings belonging to the OTHER family -- what the panel greys out. Greyed, never removed: INVARIANTS §6. A key absent from the dict makes the pipeline fall back to its own default, which can differ from the value the module needs and says nothing when it does. :param family: the chosen family. :returns: setting keys the other family owns. :raises ClassifierFamilyError: unknown family. """ key = str(family).strip().lower() if key not in FAMILY_SETTINGS: raise ClassifierFamilyError( f"{family!r} is not one of {list(CLASSIFIER_FAMILIES)}") mine = set(FAMILY_SETTINGS[key]) return tuple(k for other, keys in FAMILY_SETTINGS.items() if other != key for k in keys if k not in mine)
[docs] def resolve_ml_model_type(settings: Mapping[str, Any]) -> str: """Return the classical-ML estimator selected by ``settings``. ``model_type_ml`` is authoritative in a merged payload. ``model_type`` is accepted only when the ML-specific key is absent, which migrates the short-lived shared-vocabulary settings files written before the two model controls were separated. The default matches :func:`spacr.settings.set_default_analyze_screen`. :param settings: settings for an ML-family run. :returns: a member of :data:`ML_MODEL_TYPES`. :raises ValueError: when the selected value is not an ML estimator. """ selected = settings.get("model_type_ml") if selected in (None, ""): selected = settings.get("model_type", "xgboost") model_type = str(selected).strip().lower() if model_type not in ML_MODEL_TYPES: raise ValueError( f"Unsupported model_type_ml: {selected!r}. Choose one of " f"{list(ML_MODEL_TYPES)}") return model_type
[docs] def classify(settings: Mapping[str, Any]) -> Any: """Run whichever classifier family ``settings`` asks for. The merged module's pipeline entry point. It normalises the shared vocabulary, resolves the family, and calls the existing entry point unchanged -- so a run here and a run through Classify (CV) or Classify (ML) are the same run, not two implementations that have to be kept in step. :param settings: the run settings. :returns: whatever the dispatched pipeline returns. :raises ClassifierFamilyError: an unrecognised family. :raises ValueError: when pre-dispatch validation rejects the selected ML estimator or CV crop source. """ from .classify_classes import normalize_settings as normalize_classes from .training_basis import normalize_settings family = resolve_family(settings) ml_model_type = ( resolve_ml_model_type(settings) if family == "ml" else None ) try: _begin_flowview_run(settings) except Exception: pass resolved = dict(normalize_classes(normalize_settings(settings))) if family == "cv": from .crop_source import validate as validate_crops validate_crops(resolved) if family == "ml": from .ml import generate_ml_scores resolved["model_type_ml"] = ml_model_type if "test_split" in resolved: resolved.setdefault("test_size", resolved["test_split"]) if "cross_validation_enabled" in resolved: resolved.setdefault("cross_validation", resolved["cross_validation_enabled"]) return generate_ml_scores(resolved) from .deep_spacr import deep_spacr return deep_spacr(resolved)