"""One Classify entry point over both classifier families.
**What it is for.** Classify is the fourth step of the pipeline, after Mask,
Measure and Annotate. It learns the classes a screen is scored on and
predicts them for every object. ``classifier_family`` chooses how: ``cv``
trains an image model on object crops through
:func:`spacr.deep_spacr.deep_spacr`, and ``ml`` fits a classical model such as
XGBoost, LightGBM or a random forest on measured features through
:func:`spacr.ml.generate_ml_scores`.
**What it needs.** A project that has been through Mask and Measure, so that
``measurements/measurements.db`` and its object crops exist, and a definition
of the classes. ``dataset_mode`` sets that definition for both families:
``metadata`` takes classes from plate metadata such as the positive and
negative control wells, and ``annotation`` takes them from an annotation
column of ``png_list``, as written by Annotate. The CV family reads crops from
PNG files or cuts them from the ``merged`` arrays, as ``crop_source`` selects.
**What it produces.** A CV run writes model checkpoints,
``DL_model_settings.csv``, a dataset tar, a ``top_examples/`` folder with the
most confident crops of each class and an evaluation bundle with the held-out
performance and ``leakage.json``, and merges each object's predicted class and
probability into ``png_list``. An ML run writes per-object predictions,
feature-importance and permutation tables and a plate heatmap to
``results/`` beside the measurements database.
**What to do next.** Check the held-out performance, and the leakage verdict
on the QC screen, before trusting the predictions. Then take per-well scores
into Regression, which pairs them with the guide counts from Map Barcodes to
estimate which guides and genes explain the phenotype.
Classify (CV) trains a Torch model on object crops. Classify (ML) fits a
gradient-boosted model on measured features. They answer the same question --
*which class is this object* -- and until now they were two modules that
shared six setting names out of 78 and 37, two of which disagreed on their
default.
:mod:`spacr.training_basis` already unified what defines a CLASS. This module
unifies what runs: one settings dict, one ``classifier_family`` switch, one
call. The two original modules stay exactly as they are -- a merged screen
that removed them would strand every saved settings CSV and every notebook
that imports their entry points.
**Nothing here reimplements either pipeline.** ``deep_spacr`` and
``generate_ml_scores`` are called unchanged, which is what makes the merged
module honest: a run through it and a run through the module it dispatches to
produce the same result, because they are the same code.
"""
from __future__ import annotations
import os
import sys
from typing import Any, Dict, Mapping, Tuple
#: The two classifier families, in the order the settings panel offers them.
CLASSIFIER_FAMILIES: Tuple[str, ...] = ("cv", "ml")
ML_MODEL_TYPES: Tuple[str, ...] = (
"xgboost", "lightgbm", "catboost", "random_forest", "extra_trees",
"gradient_boosting", "logistic_regression", "svm", "mlp",
)
#: family -> the app key whose settings and pipeline it uses. The merged
#: screen is a front end onto these, not a replacement for them.
FAMILY_APP_KEY: Dict[str, str] = {"cv": "classify", "ml": "ml_analyze"}
#: Settings each family reads that the other has no use for. Drives the
#: greying, the same way :data:`spacr.training_basis.BASIS_SETTINGS` does for
#: the training basis, and for the same reason: a control the user can edit
#: that changes nothing is worse than one that is not there.
FAMILY_SETTINGS: Dict[str, Tuple[str, ...]] = {
"cv": (
"crop_shape", "extract_channels", "object_array", "coordinate_columns",
"model_type", "custom_model_path", "image_size",
"train_channels", "epochs", "optimizer_type", "schedule", "loss_type",
"dropout_rate", "init_weights", "amsgrad", "weight_decay",
"gradient_accumulation_steps",
"early_stopping_patience", "augment", "pin_memory", "use_checkpoint",
"resume_checkpoint", "tensorboard", "focal_gamma", "focal_alpha",
"label_smoothing", "logit_adjust_tau", "train", "test",
"generate_training_dataset", "apply_model_to_dataset",
"generate_full_dataset", "tar_path", "n_top_examples", "path_string",
"file_type", "crop_source", "batch_size", "val_split",
"image_source", "mixed_precision",
),
"ml": (
"model_type_ml", "n_estimators", "reg_alpha", "reg_lambda",
"prune_features", "top_features", "n_repeats", "min_cells_per_well",
"remove_low_variance_features", "remove_highly_correlated_features",
"heatmap_feature", "grouping", "min_max", "cmap",
"batch_correction", "batch_column", "batch_control_column",
"batch_control_values", "batch_covariate_column",
"batch_combat_mean_only", "batch_min_samples", "batch_missing_control",
"nuclei_limit", "pathogen_limit", "exclude",
),
}
[docs]
class ClassifierFamilyError(ValueError):
"""A classifier family spaCR does not have."""
def _begin_flowview_run(settings: Mapping[str, Any]) -> object | None:
"""Start a live graph only when optional FlowView tracing is enabled.
:param settings: Raw Classify settings fingerprinted into the new
FlowView graph.
:returns: The newly installed collector when tracing is enabled,
otherwise ``None``.
The common disabled path is a module-cache lookup and an environment
check; importantly, it imports no FlowView code. A panel can enable the
already-loaded trace module, while ``SPACR_FLOWVIEW`` opts a headless run
in through the same lazy boundary.
"""
trace_module = sys.modules.get("spacr.flowview.trace")
if trace_module is None:
enabled_by_environment = os.environ.get("SPACR_FLOWVIEW", "")
if enabled_by_environment.strip().casefold() not in {
"1",
"on",
"true",
"yes",
}:
return None
from .flowview import trace as trace_module
if not trace_module.is_enabled():
return None
from .flowview.classify_blueprint import _install_classify_collector
return _install_classify_collector(settings)
[docs]
def resolve_family(settings: Mapping[str, Any]) -> str:
"""Return the classifier family a settings dict asks for.
Defaults to ``'cv'``, because the merged module's own default settings
are the CV ones and a dict with no family is most likely a Classify (CV)
CSV opened in the merged screen.
:param settings: the run settings.
:returns: ``'cv'`` or ``'ml'``.
:raises ClassifierFamilyError: an unrecognised family. Guessing would
train a different kind of model than the user asked for and report
success.
"""
declared = settings.get("classifier_family")
if declared is None or declared == "":
return "cv"
family = str(declared).strip().lower()
if family not in CLASSIFIER_FAMILIES:
raise ClassifierFamilyError(
f"classifier_family={declared!r} is not one of "
f"{list(CLASSIFIER_FAMILIES)}")
return family
[docs]
def inapplicable_settings(family: str) -> Tuple[str, ...]:
"""Settings belonging to the OTHER family -- what the panel greys out.
Greyed, never removed: INVARIANTS §6. A key absent from the dict makes
the pipeline fall back to its own default, which can differ from the
value the module needs and says nothing when it does.
:param family: the chosen family.
:returns: setting keys the other family owns.
:raises ClassifierFamilyError: unknown family.
"""
key = str(family).strip().lower()
if key not in FAMILY_SETTINGS:
raise ClassifierFamilyError(
f"{family!r} is not one of {list(CLASSIFIER_FAMILIES)}")
mine = set(FAMILY_SETTINGS[key])
return tuple(k for other, keys in FAMILY_SETTINGS.items()
if other != key for k in keys if k not in mine)
[docs]
def resolve_ml_model_type(settings: Mapping[str, Any]) -> str:
"""Return the classical-ML estimator selected by ``settings``.
``model_type_ml`` is authoritative in a merged payload. ``model_type``
is accepted only when the ML-specific key is absent, which migrates the
short-lived shared-vocabulary settings files written before the two model
controls were separated. The default matches
:func:`spacr.settings.set_default_analyze_screen`.
:param settings: settings for an ML-family run.
:returns: a member of :data:`ML_MODEL_TYPES`.
:raises ValueError: when the selected value is not an ML estimator.
"""
selected = settings.get("model_type_ml")
if selected in (None, ""):
selected = settings.get("model_type", "xgboost")
model_type = str(selected).strip().lower()
if model_type not in ML_MODEL_TYPES:
raise ValueError(
f"Unsupported model_type_ml: {selected!r}. Choose one of "
f"{list(ML_MODEL_TYPES)}")
return model_type
[docs]
def classify(settings: Mapping[str, Any]) -> Any:
"""Run whichever classifier family ``settings`` asks for.
The merged module's pipeline entry point. It normalises the shared
vocabulary, resolves the family, and calls the existing entry point
unchanged -- so a run here and a run through Classify (CV) or Classify
(ML) are the same run, not two implementations that have to be kept in
step.
:param settings: the run settings.
:returns: whatever the dispatched pipeline returns.
:raises ClassifierFamilyError: an unrecognised family.
:raises ValueError: when pre-dispatch validation rejects the selected ML
estimator or CV crop source.
"""
from .classify_classes import normalize_settings as normalize_classes
from .training_basis import normalize_settings
family = resolve_family(settings)
ml_model_type = (
resolve_ml_model_type(settings) if family == "ml" else None
)
try:
_begin_flowview_run(settings)
except Exception:
pass
resolved = dict(normalize_classes(normalize_settings(settings)))
if family == "cv":
from .crop_source import validate as validate_crops
validate_crops(resolved)
if family == "ml":
from .ml import generate_ml_scores
resolved["model_type_ml"] = ml_model_type
if "test_split" in resolved:
resolved.setdefault("test_size", resolved["test_split"])
if "cross_validation_enabled" in resolved:
resolved.setdefault("cross_validation",
resolved["cross_validation_enabled"])
return generate_ml_scores(resolved)
from .deep_spacr import deep_spacr
return deep_spacr(resolved)