Source code for spacr.ml

"""Find screen phenotypes and test which guides or genes explain them.

WHAT IT IS FOR
==============
The **Regression** tile opens this module because
:func:`perform_regression` is its main entry point: it joins per-well image
scores to sequencing counts and estimates guide- or gene-level associations
for a pooled screen.  The module also contains a separate classical
machine-learning workflow, :func:`generate_ml_scores`, which trains a
classifier on measured single-object features and turns those predictions
into the score table a regression can consume.

WHAT IT NEEDS
=============
A regression run is configured with ``paired_data``: ordered score/count CSV
pairs whose plate and well identities agree.  Older ``score_data`` and
``count_data`` lists are migrated positionally, but explicit pairs are safer.
Choose the score column with ``dependent_variable`` and provide the relevant
control wells, plate metadata, analysis ``level`` (guide, gene, or both),
multiple-testing threshold, and either a supported ``regression_type`` or
``None`` for distribution-based selection.  Family-specific settings are
validated rather than silently ignored.  The classical-ML path instead needs
one or more ``measurements.db`` files, labelled positive and negative controls
or an annotation column, and a model choice such as XGBoost, logistic
regression, or random forest.

WHAT IT PRODUCES
================
Regression results go into a new, non-overwriting
``<output root>/results/<analysis kind>[_n]`` directory.  ``results.csv`` is
the primary combined table; ``results_grna.csv`` and ``results_gene.csv``
make the fitted levels explicit, and ``results_significant.csv`` records
thresholded hits.  The same directory holds summaries, diagnostics, volcano
and plate figures, optional publication-panel packages, resource measurements,
or a detailed failure report.  A requested level with no fitted rows is kept
as a header-only CSV so downstream tools can distinguish "tested, no rows"
from a missing artifact.  Classical ML writes predictions, feature-importance
tables, evaluation results, and a plate heatmap beneath ``results``.

WHAT TO DO NEXT
===============
Read the diagnostic and failure/resource records before ranking hits, then
review the significant table alongside the complete level tables and check
whether guide and gene effects agree.  Open the generated result panels for
visual QC and retain the settings/manifests with any reported hit list.  If
scores do not yet exist, run :func:`generate_ml_scores`; if counts do not yet
exist, create them with
:func:`spacr.sequencing.generate_barecode_mapping`.

Several statistical distinctions are deliberate.  Guide and gene fits are
separate multiple-testing families and receive separate corrections; the
nominal ``alpha``, effect-size threshold, and corrected significance cutoff
are not interchangeable.  A mixed model reports guide effects as shrunken
BLUP predictions without guide p- or q-values, so they must not be read as a
second guide significance test.  Diagnostic or report-generation failures do
not erase successful scientific output, while an actual regression failure
is recorded and then re-raised unchanged so callers cannot mistake it for a
completed run.
"""

import functools
import logging
import os, sys, re
import pandas as pd
import numpy as np
from scipy import stats
from scipy.stats import shapiro
from math import pi

from sklearn.linear_model import (Lasso, Ridge, LassoCV, RidgeCV,
                                  ElasticNet, ElasticNetCV)
from sklearn.svm import LinearSVC
from sklearn.base import clone
from sklearn.metrics import mean_squared_error

import matplotlib.pyplot as plt
try:
    from IPython.display import display
except Exception:
[docs] def display(*args, **kwargs): """Accept and discard display arguments when IPython is unavailable.""" pass
import scipy.stats as st import statsmodels.api as sm import statsmodels.formula.api as smf from statsmodels.regression.mixed_linear_model import MixedLM from statsmodels.stats.outliers_influence import variance_inflation_factor from statsmodels.genmod.families import Binomial from statsmodels.genmod.families.links import Logit from statsmodels.othermod.betareg import BetaModel from sklearn.preprocessing import FunctionTransformer from patsy import dmatrices from .regression_spec import (DEFAULT_REGRESSION_BACKEND, # noqa: F401 NO_P_VALUE_TYPES, REGRESSION_BACKENDS, REGRESSION_BACKEND_ORDER, REGRESSION_SETTINGS_USED, REGRESSION_TYPES, RUN_LEVEL_SETTINGS, UNSUPPORTED_REGRESSION_TYPES, _MODEL_LEVEL_DEFAULTS, _RUN_LEVEL_DEFAULTS) from .regression_families import (REGRESSION_FAMILY_ASSUMPTIONS, # noqa: F401 REGRESSION_FAMILY_GROUPS, family_group, family_label, regression_family_choices) from .mixed_gpu import MixedBackendUnavailable # noqa: F401 from .regression_backends import (backend_label, # noqa: F401 backend_status, backend_supports, resolve_backend_name) from sklearn.model_selection import StratifiedKFold from sklearn.feature_selection import SelectKBest, f_classif from sklearn.ensemble import RandomForestClassifier, HistGradientBoostingClassifier from sklearn.linear_model import LogisticRegression from sklearn.inspection import permutation_importance from sklearn.metrics import classification_report, precision_recall_curve from sklearn.preprocessing import StandardScaler from sklearn.preprocessing import MinMaxScaler from scipy.spatial.distance import cosine, euclidean, mahalanobis, cityblock, minkowski, chebyshev, braycurtis from xgboost import XGBClassifier from . import frame_handoff, schema, tabular from .openmp_guard import single_threaded_openmp, guarded_n_jobs from .plot import save_figure LOG = logging.getLogger("spacr.ml") _FLOWVIEW_TRUE_VALUES = frozenset({"1", "on", "true", "yes"}) def _flowview_event(action, *args): """Reach optional Classify tracing without importing it when disabled.""" trace_module = sys.modules.get("spacr.flowview.trace") if trace_module is None: enabled_by_environment = os.environ.get("SPACR_FLOWVIEW", "") if enabled_by_environment.strip().casefold() not in _FLOWVIEW_TRUE_VALUES: return False try: from .flowview import trace as trace_module except BaseException: return False try: if not trace_module.is_enabled(): return False from .flowview import _classify_stages return bool(getattr(_classify_stages, f"_{action}")(*args)) except BaseException: return False def _flowview_pipeline(family): """Finish or fail the active graph without changing scientific output.""" def decorate(function): """Return a metadata-preserving lifecycle wrapper for ``function``.""" @functools.wraps(function) def observed(*args, **kwargs): """Call unchanged, reporting success or failure to an active trace.""" settings = args[0] if args else kwargs.get("settings") active = _flowview_event("begin", settings, family) try: result = function(*args, **kwargs) except BaseException as scientific_error: if active: _flowview_event("fail", scientific_error) raise if active: _flowview_event("finish") return result return observed return decorate def _flowview_advance(node_id): """Record one real operation boundary, or do nothing when disabled.""" _flowview_event("advance", node_id) def _flowview_metric(name, value): """Record one scalar on the active stage, or do nothing when disabled.""" _flowview_event("metric", name, value) from scipy.stats import kstest, normaltest import matplotlib from .figures.style import ROLES, figure_style, theme_target if not (sys.platform.startswith(('win', 'darwin')) or os.environ.get('DISPLAY')): matplotlib.use('Agg') import warnings def _require_backend(regression_type, regression_backend): """Resolve and validate a regression backend for a model family. Validate at run time because settings loaded from a file can bypass the GUI's disabled backend entries. Never substitute another backend silently. :param regression_type: the family being fitted. :param regression_backend: name or label, or ``None`` for the default. :returns: the canonical backend name. :raises ValueError: when the backend cannot fit the family, is not installed, or needs a GPU that is not there. """ name = resolve_backend_name(regression_backend) if name == DEFAULT_REGRESSION_BACKEND: return name status = backend_status(name, regression_type) if not status['enabled']: raise ValueError( f"{status['reason']} Set regression_backend='statsmodels' to fit " f"it with the default backend, which produced every existing " f"result.") return name def _say_what_a_mixed_fit_will_cost(backend, df=None): """Describe the expected cost of a statsmodels mixed fit before it starts. Print nothing for non-default backends. For statsmodels, include the row count when available and state whether the compatible Torch GPU backend can be selected instead. """ if backend != DEFAULT_REGRESSION_BACKEND: return rows = None try: rows = len(df) if df is not None else None except TypeError: # noqa: BLE001 rows = None try: status = backend_status('torch', 'mixed') available = bool(status.get('enabled')) reason = str(status.get('reason') or '') except Exception: # noqa: BLE001 available, reason = False, '' size = f" on {rows} wells" if rows else "" print(f"Fitting the mixed model{size} with statsmodels. This is the slow " f"one: dense linear algebra, measured at 54x OLS on 40 genes and " f"rising with screen size, and it prints nothing while it runs.") if available: print(" The same model, same estimates, is available on the GPU: set " "regression_backend='torch'. Measured on this screen: 26 " "seconds against >25 minutes.") elif reason: print(f" The GPU backend would be faster but is not usable here: " f"{reason}") #: Settings that must lie strictly inside 0 and 1, and what each one is for. #: Checked BEFORE the run writes anything -- see #: :func:`_reject_impossible_probabilities`. _UNIT_INTERVAL_SETTINGS = { 'fdr_alpha': "the family-level rejection threshold for adjusted P values", 'p_threshold_alpha': "the cut applied to the P value column", 'alpha': None, } def _reject_impossible_probabilities(settings): """Validate probability thresholds before the run writes output. Check every setting in :data:`_UNIT_INTERVAL_SETTINGS` that represents a probability and reject non-numeric values or values outside the open interval ``(0, 1)``. The penalty parameter named ``alpha`` is deliberately excluded because it is not a probability. """ for key, what in _UNIT_INTERVAL_SETTINGS.items(): if what is None or key not in settings: continue value = settings.get(key) if value is None: continue try: number = float(value) except (TypeError, ValueError): raise ValueError( f"{key}={value!r} is not a number. It is {what}, and must " f"be strictly between 0 and 1 (usually 0.05).") from None if not (0.0 < number < 1.0): raise ValueError( f"{key}={number!r} is outside 0 and 1. It is {what}, so it " f"has no meaning there; the usual value is 0.05. " + ("A value one less than what you meant is what a spin box " "does when its arrow or the scroll wheel is nudged -- " f"{number + 1:g} may be the number you set." if -1.0 < number < 0.0 else "Set it in the Significance section, or in " "settings/regression.csv if this run came from a file.")) def _concat_named_csvs(paths): """Read every CSV in ``paths`` into one frame. A screen's counts and scores are one file per plate, and this question is asked of the screen rather than of a plate, so they are read together. Each file goes through :func:`spacr.tabular.read_table`, so a header spelled ``column_name`` or ``Well`` reaches the fit under the canonical key names. :param paths: one path, a list of them, or nothing. :returns: the concatenated frame. :raises ValueError: when there is nothing readable to concatenate. """ import pandas as pd from .tabular import read_table if not paths: raise ValueError("no table was given to read") if isinstance(paths, (str, os.PathLike)): paths = [paths] frames = [] for one in paths: try: frames.append(read_table(one)) except Exception as exc: raise ValueError(f"{one} could not be read: {exc}") from exc if not frames: raise ValueError("no table was given to read") return pd.concat(frames, ignore_index=True) def _well_block_tokens(settings, key): """The row/column/well tokens one control-block setting names. :param settings: the run's settings mapping. :param key: ``'positive_control_wells'`` or its negative twin. :returns: the tokens, lower-cased, with the empty ones dropped. A list or a bare string, because both spellings reach here: the panel writes a list and a settings CSV can carry either. """ raw = (settings or {}).get(key) if raw is None: return [] values = raw if isinstance(raw, (list, tuple, set)) else [raw] return [str(value).strip().lower() for value in values if str(value).strip()] def _wells_in_block(labels, tokens): """Which of ``labels`` the block ``tokens`` name. :param labels: well labels as the score table carries them, ``prc`` form -- ``plate_row_column``. :param tokens: what the plate design calls the block: a column (``c2``), a row (``r1``) or a whole well (``plate1_r1_c2``). :returns: the matching labels, sorted and de-duplicated. THE TOKEN IS MATCHED AGAINST A PART, NOT AS A SUBSTRING. ``'c2' in 'plate1_r1_c20'`` is true and says nothing, so a plate wider than nine columns would fold column 20 into column 2's reference and shift the endpoint the whole calibration is anchored on. """ wanted = set(tokens) out = set() for label in labels: text = str(label).strip() parts = {part.strip().lower() for part in text.split('_')} parts.add(text.lower()) if parts & wanted: out.add(text) return sorted(out) def _calibration_inputs(settings): """Gather what the fraction-threshold sweep needs, from the run's own files. THE IMAGING SIDE IS THE CLASSIFIER SCORE. `mixed_ratio_calibration` takes an ``(n_cells, n_features)`` block and one well label per cell, and asks what mixture of positive and negative control each well looks like. The per-cell score column is exactly that measurement with one feature, so the score table the run already loaded is the imaging side and no second source is needed. THE PURE WELLS ARE NAMED FROM THE PLATE DESIGN, through `positive_control_wells` and `negative_control_wells`. Identifying them by their reported fraction would be circular: that fraction is the quantity under test, and a bias large enough to matter pushes a pure well the wrong side of any cut-off. THE WELLS AND THE GUIDE ARE TWO SETTINGS, and reading one for the other is what stopped this running at all. `positive_control_id` is a gene or gRNA ID SUBSTRING in a regression -- it defaults to '239740' -- and was being matched against well labels, which no well label has ever contained. So every screen that ticked the box was refused with "no well matched", and the three control-block settings that exist to answer this were never read. The guide is what `positive_guide` needs; the wells are what `pure_pc_wells` needs. :param settings: the regression settings. :returns: keyword arguments for :func:`spacr.fraction_calibration.sweep_fraction_threshold`. :raises ValueError: when the screen cannot answer the question -- no control-well block named, no positive-control guide, or no score column to read. The caller turns that into a printed reason and the threshold the settings gave. """ import numpy as np positive_wells = _well_block_tokens(settings, 'positive_control_wells') negative_wells = _well_block_tokens(settings, 'negative_control_wells') if not positive_wells or not negative_wells: raise ValueError( "the plate design names no positive_control_wells and " "negative_control_wells, and a control-well calibration has " "nothing to calibrate against") positive_guide = str(settings.get('positive_control_id') or '').strip() if not positive_guide: raise ValueError( "positive_control_id names no gRNA, so there is no guide whose " "sequenced share can be compared with the imaging") counts = _concat_named_csvs(settings.get('count_data')) scores = _concat_named_csvs(settings.get('score_data')) well_column = str(settings.get('count_well_column') or 'prc') score_column = str(settings.get('dependent_variable') or 'pred') if score_column not in scores.columns: raise ValueError( f"the score table has no {score_column!r} column to read the " f"imaging side from") if well_column not in scores.columns: raise ValueError( f"the score table has no {well_column!r} column, so a cell " f"cannot be placed in a well") usable = scores[[well_column, score_column]].dropna() features = np.asarray(usable[score_column], dtype=float).reshape(-1, 1) wells = [str(w) for w in usable[well_column]] pure_pc = _wells_in_block(wells, positive_wells) pure_nc = _wells_in_block(wells, negative_wells) if not pure_pc or not pure_nc: raise ValueError( f"no well matched {positive_wells} and {negative_wells}, so " f"there is no pure control to anchor the fit") return { "counts": counts, "features": features, "wells": wells, "positive_guide": positive_guide, "pure_pc_wells": pure_pc, "pure_nc_wells": pure_nc, "normalise": bool(settings.get('normalise_fraction', True)), "well_column": well_column, "guide_column": str(settings.get('count_grna_column') or 'grna'), "count_column": str(settings.get('count_value_column') or 'count'), } def _calibrated_fraction_threshold(settings): """The cut-off the control wells imply, or ``None`` if they cannot say. Returns ``None`` -- rather than raising -- for every reason the sweep might not apply: the plate design names no pure control wells, there are too few of them to fit anything, the counts are missing the columns it reads, or the optional module is not importable. Each of those is an ordinary answer to "can this screen calibrate itself", and none of them is a reason to stop a run that already had a usable threshold. :param settings: the regression settings, read for the control-well names and the count table. :returns: the measured threshold, or ``None``. """ try: from .fraction_calibration import sweep_fraction_threshold except Exception: print("fraction-threshold calibration is unavailable; " "using the threshold as given") return None try: result = sweep_fraction_threshold(**_calibration_inputs(settings)) except (KeyError, ValueError, TypeError) as exc: print(f"fraction-threshold calibration did not run: {exc}") return None chosen = result.get("chosen") if isinstance(result, dict) else None if chosen is None: print("fraction-threshold calibration found no cut-off it preferred; " "using the threshold as given") return None try: from .fraction_calibration import describe print(describe(result)) except Exception: print(f"fraction_threshold calibrated to {chosen}") return float(chosen) def _graph_sequencing_stats(settings): """Resolve the sequencing threshold helper through one testable seam.""" from .sequencing import graph_sequencing_stats return graph_sequencing_stats(settings) #: File types the run treats as a figure when it collects what a helper drew. _FIGURE_SUFFIXES = ('.pdf', '.png', '.svg', '.jpg', '.jpeg', '.tif', '.tiff', '.eps') def _screen_figure_folders(settings): """Where a sequencing helper drops its figures: beside the COUNT DATA. `graph_sequencing_stats` writes ``<count folder>/results/`` for the threshold sweep and ``<count folder>/`` for the unique-count plate heatmap, both derived from ``settings['count_data'][0]`` inside :mod:`spacr.sequencing`. Neither is the run's own folder, which is the whole problem this list exists to solve. """ folders = [] for path in (settings.get('count_data') or []): base = os.path.dirname(str(path)) for candidate in (base, os.path.join(base, 'results')): if candidate and candidate not in folders: folders.append(candidate) return folders def _figure_stamps(folders): """``{path: (mtime, size)}`` for every figure directly inside ``folders``. Not recursive, and not a bare listing: a run of the same screen writes the same file NAMES, so identity has to include the stamp or a figure left by yesterday's run reads as one this run drew. """ stamps = {} for folder in folders: try: entries = list(os.scandir(folder)) except OSError: continue for entry in entries: if not entry.name.lower().endswith(_FIGURE_SUFFIXES): continue try: if not entry.is_file(): continue info = entry.stat() except OSError: continue stamps[entry.path] = (info.st_mtime_ns, info.st_size) return stamps def _keep_figures_with_the_run(before, folders, destination): """Copy newly written figures into the run-specific output folder. Compare current file stamps with ``before`` and copy only new or changed figures. Retain the originals because other workflows may reference the screen-level folder. :param before: the stamps from :func:`_figure_stamps` taken first. :param folders: the same folders it was taken over. :param destination: the run folder. :returns: the paths written, so the caller can name them. """ import shutil kept = [] for path, stamp in sorted(_figure_stamps(folders).items()): if before.get(path) == stamp: continue target = os.path.join(destination, os.path.basename(path)) if os.path.abspath(target) == os.path.abspath(path): continue try: os.makedirs(destination, exist_ok=True) shutil.copy2(path, target) except OSError as error: print(f"Could not keep {os.path.basename(path)} with the run: " f"{error}") continue kept.append(target) return kept def _run_random_state(default=None): """Return the active run's seed, for an estimator's ``random_state=``. Imported inside the call rather than at module scope: :mod:`spacr.runctx` reaches :mod:`spacr.settings`, which reaches back here, and a top-level import would be a cycle. Outside a run this is whatever ``default`` was, which is the literal these call sites used to hard-code. :param default: the value to use when no run is open. :returns: the run seed, or ``default``. """ from .runctx import random_state return random_state(default) warnings.filterwarnings("ignore", message="3D stack used, but stitch_threshold=0 and do_3D=False, so masks are made per plane only") class _DispersedVariance: """Scale a statsmodels variance function by a constant dispersion factor. ``Binomial.__init__`` stores a ``varfuncs`` callable in the *instance* ``__dict__`` under the name ``variance``, and an instance attribute always wins over a subclass method of the same name. Overriding ``variance`` in a subclass therefore has no effect on anything statsmodels does. Wrapping the stored callable is the only way to make the factor reach the fit, and delegating attribute lookups keeps ``family.variance.deriv`` — which ``GLM`` calls — working. :param varfunc: The variance callable installed by statsmodels. :param dispersion: Multiplicative variance scaling. """ def __init__(self, varfunc, dispersion): """Store the variance callable and its multiplicative dispersion.""" self._varfunc = varfunc self.dispersion = dispersion def __call__(self, mu): """Return ``dispersion * varfunc(mu)``.""" return self.dispersion * self._varfunc(mu) def deriv(self, mu): """Return the dispersion-scaled derivative of the variance function.""" return self.dispersion * self._varfunc.deriv(mu) def __getattr__(self, name): """Delegate every other attribute to the wrapped variance function. Raises ``AttributeError`` - never ``KeyError`` - when ``_varfunc`` is not set yet, so ``copy``/``pickle`` can probe for ``__setstate__`` and friends on a half-built instance without blowing up. """ try: varfunc = self.__dict__['_varfunc'] except KeyError: raise AttributeError(name) from None return getattr(varfunc, name)
[docs] class QuasiBinomial(Binomial): """Binomial GLM family scaled by a dispersion parameter (quasi-binomial). :param link: statsmodels link instance. Default ``Logit()``. :param dispersion: Multiplicative variance scaling. Default ``1.0``. """ def __init__(self, link=Logit(), dispersion=1.0): """Store the dispersion factor after delegating to ``Binomial``.""" super().__init__(link=link) self.dispersion = dispersion
[docs] self.variance = _DispersedVariance(self.__dict__['variance'], dispersion)
def variance(self, mu): """Adjust the variance with the dispersion parameter. :param mu: fitted mean probabilities, scalar or array; the binomial variance of ``mu`` is multiplied by ``dispersion``. """ return self.dispersion * super().variance(mu)
[docs] def calculate_p_values(X, y, model): """Return OLS-style p-values for a fitted model's coefficients. **These are not valid frequentist p-values for a penalised fit**, and the two callers that reach them know it in different ways. The standard error is the unpenalised ``rse * sqrt(diag((X'X)^-1))`` while the coefficient it is divided into has been shrunk, so the test is mis-specified. The direction of the error is the one that matters here and it is the safe one: the penalty shrinks the numerator and inflates the residual in the denominator, so the statistic is too SMALL and the p-value too large. A penalised fit under-detects here; it does not manufacture hits. ``lasso`` and ``elasticnet`` do not rely on this at all — :data:`NO_P_VALUE_TYPES` routes them to a bootstrap selection frequency instead. ``ridge`` does, because it never sets a coefficient to exactly zero and so has no selection frequency to report (every feature would score 1.0), and a conservative test is a better answer than no test. ``tests/test_regression_orientation.py`` pins the null case, which is where an anticonservative version of this would show. :param X: Design matrix (``n x p``). :param y: Observed responses. :param model: Fitted estimator exposing ``predict`` and ``coef_``. :returns: 1D array of length ``p``; entries are ``NaN`` when ``n <= p + 1``. """ y_true = np.asarray(y).ravel() y_pred = np.asarray(model.predict(X)).ravel() residuals = y_true - y_pred dof = X.shape[0] - X.shape[1] - 1 if dof <= 0: return np.full(X.shape[1], np.nan) residual_std_error = np.sqrt(np.sum(residuals ** 2) / dof) XtX = X.T @ X try: XtX_inv = np.linalg.inv(np.asarray(XtX)) except np.linalg.LinAlgError: XtX_inv = np.linalg.pinv(np.asarray(XtX)) se = residual_std_error * np.sqrt(np.diag(XtX_inv)) coefs = np.asarray(model.coef_).ravel() with np.errstate(divide='ignore', invalid='ignore'): t_stats = np.where(se > 0, coefs / se, 0.0) p_values = 2 * (1 - st.norm.cdf(np.abs(t_stats))) return p_values
[docs] def perform_mixed_model(y, X, groups, alpha=None, regression_backend=DEFAULT_REGRESSION_BACKEND): """Fit a mixed-effects linear model with ``groups`` as the random intercept. Collinearity is REPORTED, never silently corrected. The previous revision reacted to any VIF above 10 by fitting .. code-block:: python ridge = Ridge(alpha=alpha).fit(X, y) X_ridge = ridge.coef_ * X # "Adjust X with Ridge coefficients" MixedLM(y, X_ridge, groups=groups) which is not ridge regression and not a mixed model of anything. It multiplies every column by that column's ridge coefficient, so * a column whose ridge coefficient is 0 - which is most of them on a screen-scale one-hot design - becomes a column of zeros, and the design is singular. That is the ``numpy.linalg.LinAlgError: Singular matrix`` that ``regression_type='mixed'`` died with on real data, thrown from inside statsmodels with nothing naming the cause; * where it did fit, every fixed effect came back multiplied by an arbitrary per-column constant, so the coefficients written to ``results.csv`` and ranked on the volcano plot were not effects on the response at all. That is the worse of the two outcomes, because it completes. A one-hot design against an intercept ALWAYS trips VIF > 10, so this path was the normal one, not the exception. :param y: Response vector. :param X: Fixed-effects design matrix (DataFrame). :param groups: Cluster identifiers for the random intercept - one entry per row of ``X``. :param alpha: Must be None. Accepted only so an old call site fails with an explanation instead of a TypeError. :param regression_backend: WHO fits it. ``'statsmodels'`` is the default and produced every existing result; ``'torch'`` fits the same profiled REML objective on the GPU (:mod:`spacr.mixed_gpu`) and returns a result object with the same attributes, so nothing downstream can tell which ran except by asking. A backend that cannot fit ``'mixed'`` here, is not installed, or needs a GPU that is absent is REFUSED with the reason -- see :func:`_require_backend`. :returns: Fitted ``statsmodels`` ``MixedLMResults``, or the equivalent :class:`spacr.mixed_gpu.TorchMixedResults`. :raises ValueError: if ``groups`` is None, if ``alpha`` is given, if ``groups`` does not align with ``X``, or if the fixed-effects design is rank-deficient (which MixedLM would otherwise report as a bare LinAlgError from three frames deep). """ if groups is None: raise ValueError("Groups must be defined for mixed model regression") if alpha is not None: raise ValueError( "perform_mixed_model takes no penalty: MixedLM has none, and the " f"alpha={alpha!r} this used to accept rescaled the design by its " "ridge coefficients, which changes what every fixed effect means. " "Drop alpha, or fit 'ridge' if you want a penalised model.") n_groups = len(np.asarray(groups).reshape(-1)) if n_groups != X.shape[0]: raise ValueError( f"groups has {n_groups} entries but the design has {X.shape[0]} " f"rows; each row must carry its own cluster id.") X_np = np.asarray(X, dtype=float) with np.errstate(divide='ignore', invalid='ignore'): vif = [variance_inflation_factor(X_np, i) for i in range(X_np.shape[1])] print(f"VIF: {vif}") if any(v > 10 for v in vif): high = [str(c) for c, v in zip(X.columns, vif) if v > 10] print(f"Multicollinearity detected with VIF > 10 for: {high}. The " f"mixed model is fitted on the design as given - the estimates " f"for those terms are unstable, not wrong. Drop or merge the " f"aliased terms, or fit 'ridge', if that matters for the " f"comparison you are making.") rank = np.linalg.matrix_rank(X_np) if rank < X_np.shape[1]: raise ValueError( f"the fixed-effects design is rank {rank} with {X_np.shape[1]} " f"columns, so its coefficients are not identified and MixedLM " f"cannot solve for them. Some terms are exact linear combinations " f"of others - typically a row/column dummy that is constant within " f"every group, or a gRNA present in exactly one well. Drop the " f"aliased terms, or use random_row_column_effects=True to move " f"the plate geometry out of the fixed effects.") backend = _require_backend('mixed', regression_backend) if backend == 'torch': from .mixed_gpu import fit_mixed_reml_torch try: fit = fit_mixed_reml_torch(y, X, groups) except Exception as exc: # noqa: BLE001 if not _is_out_of_memory(exc): raise print(f"■ The GPU ran out of memory during the mixed fit " f"({exc.__class__.__name__}). The card is shared, and what " f"was free when the design was checked was gone by the " f"time the fit asked for it. Falling back to " f"statsmodels on the CPU: same model, same numbers, " f"slower. Re-run when the card is quieter, or set " f"regression_backend='statsmodels (CPU)' to skip the " f"attempt.") try: import torch torch.cuda.empty_cache() except Exception: # noqa: BLE001 pass return MixedLM(y, X, groups=groups).fit() print(fit.summary_line()) return fit return MixedLM(y, X, groups=groups).fit()
def _is_out_of_memory(exc) -> bool: """Is this exception a device or host memory exhaustion? Matched by NAME as well as by type, because `torch.cuda.OutOfMemoryError` only exists once torch is imported and this must not import it to find out. A plain `MemoryError` counts too -- `mixed_gpu` raises one deliberately when the design will not fit. """ if isinstance(exc, MemoryError): return True name = type(exc).__name__ if "OutOfMemory" in name: return True text = str(exc).lower() return "out of memory" in text or "cuda error: out of memory" in text
[docs] def create_volcano_filename(csv_path, regression_type, alpha, dst): """Build the path this run's volcano plot will be saved to. Path construction only: nothing is read, written or created, and the ``.pdf`` in the name is not binding - :func:`spacr.plot.save_figure` rewrites the extension to whichever format the figure preference selected. :param csv_path: Source CSV. Only its basename with the last extension stripped becomes the ``<name>_volcano_plot.pdf`` stem, and only its directory is used, when ``dst`` is falsy. The file is never opened, so a path that does not exist is fine; a bare filename yields a bare relative result rather than a path under the working directory. :param regression_type: Prefixed to the filename, unless it is exactly ``'quantile'`` - then ``alpha`` is prefixed instead. ``None`` is stamped literally, giving ``None_...``: :func:`regression` calls this before :func:`check_distribution` resolves the auto-selected model, so an auto run's plot is never named for the model it actually fitted. :param alpha: Read only on the ``'quantile'`` branch; accepted and ignored for every other type, whatever its value. :func:`regression` passes the ``quantile`` setting here, not the penalty, so two quantiles of one screen cannot overwrite each other. :param dst: Output directory. Any falsy value, ``None`` and ``''`` alike, falls back to the directory of ``csv_path``. It is not created here. :returns: The joined path, which :func:`regression` hands to :func:`spacr.plot.volcano_plot` as ``save_path``. """ volcano_filename = os.path.splitext(os.path.basename(csv_path))[0] + '_volcano_plot.pdf' volcano_filename = f"{regression_type}_{volcano_filename}" if regression_type != 'quantile' else f"{alpha}_{volcano_filename}" if dst: return os.path.join(dst, volcano_filename) return os.path.join(os.path.dirname(csv_path), volcano_filename)
[docs] def scale_variables(X, y): """Min-max scale the independent (X) and dependent (y) variables to [0, 1]. Constant columns are passed through UNCHANGED. ``MinMaxScaler`` maps a column with zero range to all-zeros, and patsy's intercept is exactly such a column, so scaling a design matrix used to silently delete its intercept: statsmodels then fitted a model through the origin and still printed an ``Intercept`` row, of 0.000, in the summary. Every coefficient in that fit absorbs the mean it can no longer estimate. :param X: Design matrix (DataFrame). :param y: Response, as a 2-D array or single-column frame. :returns: ``(X_scaled, y_scaled)`` - a DataFrame with ``X``'s columns and a 2-D ``numpy`` array. Example: .. code-block:: python X = pd.DataFrame({'Intercept': 1.0, 'a': [1.0, 2.0, 3.0]}) scale_variables(X, np.array([[0.0], [1.0], [2.0]]))[0]['Intercept'] # -> 1.0, 1.0, 1.0 (not 0.0, 0.0, 0.0) """ scaler_X = MinMaxScaler() scaler_y = MinMaxScaler() X_scaled = pd.DataFrame(scaler_X.fit_transform(X), columns=X.columns) constant = X.nunique(dropna=False) <= 1 for column in X.columns[constant.values]: X_scaled[column] = np.asarray(X[column], dtype=float) y_scaled = scaler_y.fit_transform(y) return X_scaled, y_scaled
[docs] def select_glm_family(y): """Choose a ``statsmodels`` GLM family from the range and type of the response. A coarser rule than :func:`pick_glm_family_and_link`, which also sets the link: binary values give ``Binomial``, any other values inside ``[0, 1]`` give ``QuasiBinomial``, non-negative integers give ``Poisson`` and everything else ``Gaussian``. :param y: Response vector. :returns: An unfitted ``statsmodels`` family instance on its default link. """ if np.all((y == 0) | (y == 1)): print("Using Binomial family (for binary data).") return sm.families.Binomial() elif (y >= 0).all() and (y <= 1).all(): print("Using Quasi-Binomial family (for proportion data including 0 and 1).") return QuasiBinomial() elif np.all(y.astype(int) == y) and (y >= 0).all(): print("Using Poisson family (for count data).") return sm.families.Poisson() else: print("Using Gaussian family (for continuous data).") return sm.families.Gaussian()
#: The two things a fixed-effects screen model can be ABOUT, and the term each #: one regresses on. One level per fit -- see :func:`prepare_formula`. LEVEL_TERMS: dict = { 'grna': 'fraction:grna', 'gene': 'gene_fraction:gene', } #: What ``level`` may be. ``'both'`` is not a design; it is an instruction to #: fit BOTH of the above SEPARATELY and correct each within itself. LEVEL_CHOICES: tuple = ('both', 'grna', 'gene') #: Deprecated formula fragment that combines guide and gene fractions. #: #: ``check_and_clean_data`` builds ``gene_fraction`` as the SUM of the gene's #: gRNA fractions within a well, so every ``gene_fraction:gene[G]`` column is #: the sum of gene G's ``fraction:grna`` columns whenever G's guides do not #: share a well. Combining both terms therefore creates exact linear #: dependencies and a non-identifiable design. The literal remains available #: so spaCR can detect and refuse that formula explicitly. COLLINEAR_FORMULA_FRAGMENT = 'fraction:grna + gene_fraction:gene' def _level_term(level): """``LEVEL_TERMS[level]``, with the error that says why ``'both'`` is not one. :raises ValueError: for ``'both'`` (two fits, so ask for one at a time) or for anything that is not a level at all. """ key = str(level).strip().lower() if key in LEVEL_TERMS: return LEVEL_TERMS[key] if key == 'both': raise ValueError( "level='both' runs two fits, not one design, so it has no single " "formula: call prepare_formula once with level='grna' and once " "with level='gene'. Putting both terms in one design is the " "collinear model: gene_fraction is the sum of the gene's gRNA " "fractions, so the gene block is an exact linear combination of " "the gRNA block and the coefficients are not identifiable.") raise ValueError( f"level={level!r} is not a model level. Choose one of " f"{LEVEL_CHOICES!r}.") #: What the intercept of a screen regression may be asked to be. #: #: 'fitted' the model estimates it, which is what every fit did before #: this was a choice; #: 'zero' no intercept at all -- the fit passes through the origin, so #: a guide's coefficient is its whole predicted score rather #: than a departure from a baseline; #: 'control' the response is centred on the negative controls before #: fitting, so the intercept IS the control level and every #: coefficient reads as "above or below the controls"; #: 'value' the number the user gives. The response is shifted by it and #: the term is suppressed, which pins the intercept at exactly #: that value rather than estimating one near it. INTERCEPT_MODES = ("fitted", "zero", "control", "value")
[docs] def centre_on_controls(df, dependent_variable, nc): """Subtract the negative controls' median response. Returns (df, offset). THIS IS WHAT MAKES THE INTERCEPT MEAN SOMETHING. A fitted intercept is the response where every predictor is zero, which on a screen design is a well with no guide in it -- a point that does not exist. Centred on the negative controls, the intercept is the control level, and every coefficient reads directly as "this far above or below the controls". The offset is returned rather than swallowed so the caller can report it: a coefficient table whose response was shifted, with nothing saying by how much, is a table nobody can compare with another run. :param df: the long frame the fit runs on. :param dependent_variable: the response column. :param nc: the negative-control guide or gene, as the settings name it. :returns: ``(frame, offset)``. The frame is a copy when it was changed and the original when it was not; ``offset`` is 0.0 when no control row could be identified, and the caller is expected to say so. """ import numpy as _np if not nc or dependent_variable not in getattr(df, "columns", ()): return df, 0.0 wanted = str(nc).strip().lower() if not wanted: return df, 0.0 mask = None for column in ("grna", "gene", "grna_name", "gene_name"): if column not in df.columns: continue found = df[column].astype(str).str.strip().str.lower() == wanted mask = found if mask is None else (mask | found) if mask is None or not bool(mask.any()): return df, 0.0 values = _np.asarray(df.loc[mask, dependent_variable], dtype=float) values = values[_np.isfinite(values)] if not values.size: return df, 0.0 offset = float(_np.median(values)) if offset == 0.0: return df, 0.0 shifted = df.copy() shifted[dependent_variable] = ( _np.asarray(shifted[dependent_variable], dtype=float) - offset) return shifted, offset
[docs] def prepare_formula(dependent_variable, random_row_column_effects=False, block_screen=False, level='grna', model_plate_position=True, intercept='fitted'): """Build a fixed-effects formula for one screen-analysis level. Parameters ---------- dependent_variable : str Name of the response column. random_row_column_effects : bool, default=False Reserve ``plateID``, ``rowID`` and ``columnID`` for the grouping and variance-component structure in :func:`fit_mixed_model` instead of adding them as fixed effects. block_screen : bool, default=False Add ``screenID`` as a fixed effect. Use :func:`screen_is_blockable` before enabling this for user data. level : {'grna', 'gene'}, default='grna' Resolution represented by the formula. ``'grna'`` uses ``fraction:grna`` and ``'gene'`` uses ``gene_fraction:gene``. intercept : {'fitted', 'zero', 'control', 'value'}, default='fitted' What the intercept is. ``'fitted'`` estimates it. ``'zero'`` takes it out of the design, so the fit passes through the origin and a coefficient is a whole predicted score rather than a departure from a baseline. ``'control'`` keeps the term and is completed by the caller, which centres the response on the negative controls first -- the intercept is then the control level by construction. ``'value'`` suppresses the term as well, because the caller has shifted the response by a number the user gave and the intercept is pinned at exactly that number. model_plate_position : bool, default=True Include plate position in the model. With ``random_row_column_effects=False`` it is included as fixed plate, row, and column terms; with ``random_row_column_effects=True`` plate supplies the grouping variable and row/column are variance components. New application settings default to ``False`` even though this helper retains ``True`` for API compatibility. Returns ------- str A patsy-compatible formula for one analysis level. Raises ------ ValueError If ``level`` is unknown or ``'both'``, or if random plate-position effects are requested while plate position is disabled. Notes ----- Guide and gene effects are fitted separately because the gene fraction is derived from its guide fractions; including both blocks in one design is rank deficient. Use :func:`regression_levels` to request both fits. """ from .schema import SCREEN_KEY term = _level_term(level) screen = f' + {SCREEN_KEY}' if block_screen else '' mode = str(intercept or "fitted").strip().lower() if mode not in INTERCEPT_MODES: raise ValueError( f"intercept={intercept!r} is not one of {list(INTERCEPT_MODES)}. " f"'fitted' estimates it, 'zero' fits through the origin, " f"'control' centres the response on the negative controls so " f"the intercept is the control level, and 'value' pins it at a " f"number you give.") origin = ' - 1' if mode in ('zero', 'value') else '' if random_row_column_effects and not model_plate_position: raise ValueError( "model_plate_position=False takes plateID, rowID and columnID out of the " "model entirely and random_row_column_effects=True asks for them " "as variance components, so there is no term left for the mixed " "fit to make random: one of the two has to go. Plate position " "has three states -- OUT (model_plate_position=False), FIXED " "(model_plate_position=True) and RANDOM (both True) -- and this " "is a fourth. Set model_plate_position=True to fit plate, row " "and column in the mixed-model structure, or " "random_row_column_effects=False to leave plate position out of " "the model.") if not model_plate_position: return f'{dependent_variable} ~ {term}{screen}{origin}' if random_row_column_effects: return f'{dependent_variable} ~ {term}{screen}{origin}' return (f'{dependent_variable} ~ {term} + plateID + rowID + ' f'columnID{screen}{origin}')
[docs] def screen_is_blockable(df) -> bool: """Whether ``screenID`` can be a term in this frame's design. True only when the column exists and carries more than one distinct value. A single-screen project is the normal case and must be untouched by the design: it has no screenID at all, or one value, and either way the term would be a constant column. The same rule :func:`spacr.measurement_scan._dummy_block` applies, stated once for the formula path so a frame cannot be blocked on by one and not the other. :param df: the design DataFrame, or ``None`` (returns ``False``); its ``screenID`` column is compared as strings. """ from .schema import SCREEN_KEY if df is None or SCREEN_KEY not in getattr(df, 'columns', ()): return False return int(df[SCREEN_KEY].astype(str).nunique(dropna=True)) > 1
#: How a guide's BLUP is named in the coefficient table. NOT ``fraction:grna`` #: -- a BLUP is a shrunken prediction of a random effect, not a fixed #: coefficient, and giving it the fixed term's name is exactly how it would end #: up in a hit list with a q value beside it. BLUP_FEATURE_TEMPLATE = 'blup:grna[{}]' #: What each row of a mixed fit's coefficient table IS. The column exists so #: nothing downstream has to guess from the name, and so a variance component #: or a BLUP can never be read as an effect on the response. TERM_FIXED = 'fixed' TERM_VARIANCE = 'variance' TERM_BLUP = 'random_effect_blup' def _blup_guide_name(key): """The guide id inside a statsmodels variance-component BLUP key. ``vc_formula={'grna': '0 + C(grna)'}`` labels its columns ``grna[C(grna)[244480_3]]``, so the id is the innermost bracket. Returns ``None`` for the group's own intercept (``'Group'``) and for anything that is not a guide component. """ text = str(key) match = re.search(r'C\(grna\)\[(?:T\.)?([^\]]+)\]', text) if match: return match.group(1) return None def _answering_stop(model): """Add a cancellation checkpoint to a statsmodels model instance. Wrap the instance's ``loglike`` method because optimizers evaluate it on each step, providing finer cancellation granularity than the fit callback. The wrapper propagates :class:`spacr.cancellation.PipelineCancelled` and returns the same model instance. """ from .cancellation import checkpoint original = model.loglike def loglike(*args, **kwargs): """Check for cancellation, then return the original likelihood result.""" checkpoint() return original(*args, **kwargs) model.loglike = loglike return model
[docs] def fit_mixed_model(df, formula, dst, *, random_row_column_effects=False, gene_column='gene', guide_column='grna', regression_backend=DEFAULT_REGRESSION_BACKEND): """Fit a mixed model with guides nested within genes. The model treats genes as fixed effects and guides as random effects nested within genes. In statsmodels notation, ``groups=gene`` supplies the outer random intercept and ``vc_formula={'grna': '0 + C(grna)'}`` supplies the guide-within-gene variance component. A blockable ``screenID`` supplied by :func:`prepare_formula` remains a fixed effect. With only two screen levels, a random screen variance would be estimated from one degree of freedom. The plate is not nested within the screen because plate position is already represented by the row and column structure. Single-screen data omit the constant screen term to avoid a rank-deficient design. Parameters ---------- df : pandas.DataFrame Model data containing the formula variables and the gene and guide grouping columns. formula : str Fixed-effects formula, normally returned by :func:`prepare_formula` with ``level='gene'``. dst : path-like Destination for the residual histogram. random_row_column_effects : bool, default False Add row and column variance components instead of fixed terms. gene_column : str, default 'gene' Column containing the outer gene groups. guide_column : str, default 'grna' Column containing guides nested within each gene. regression_backend : {'statsmodels', 'torch'}, default 'statsmodels' Mixed-model backend. The torch backend fits the same nested model with GPU acceleration when available. Returns ------- mixed_model Fitted backend-specific mixed-model result. coef_df : pandas.DataFrame Fixed effects, variance components, and guide BLUPs. Variance components and BLUPs have ``NaN`` p-values because they are not fixed-effect hypothesis tests. Raises ------ ValueError If required grouping columns are missing, no gene has multiple guides, or the backend cannot fit the nested design. MixedBackendUnavailable If the selected mixed-model backend is unavailable. """ from .plot import plot_histogram for column in (gene_column, guide_column): if column not in df.columns: raise ValueError( f"the mixed model nests {guide_column!r} inside " f"{gene_column!r}, and this frame has no {column!r} column. " f"Columns: {sorted(df.columns)[:20]}") response = str(formula).split('~', 1)[0].strip() or 'the response' groups = _mixed_model_groups(df, response, df.index, gene_column=gene_column) guides_per_gene = df.groupby(gene_column, observed=True)[ guide_column].nunique() if int((guides_per_gene > 1).sum()) == 0: raise ValueError( f"the mixed model nests guides inside genes, and no gene in this " f"frame has more than one guide ({len(guides_per_gene)} genes, " f"one guide each). The guide variance component would be exactly " f"confounded with the residual and would come back as zero. Use a " f"fixed-effects regression_type with level='gene' -- with one " f"guide per gene the two levels are the same model anyway.") vc_formula = {guide_column: f'0 + C({guide_column})'} if random_row_column_effects: vc_formula['rowID'] = '0 + C(rowID)' vc_formula['columnID'] = '0 + C(columnID)' backend = _require_backend('mixed', regression_backend) _say_what_a_mixed_fit_will_cost(backend, df) try: if backend == 'torch': from .mixed_gpu import mixedlm_torch mixed_model = mixedlm_torch(formula, df, groups, vc_formula=vc_formula) print(mixed_model.summary_line()) else: model = smf.mixedlm(formula, data=df, groups=groups, re_formula='1', vc_formula=vc_formula) mixed_model = _answering_stop(model).fit() except MixedBackendUnavailable: raise except Exception as error: raise ValueError( f"MixedLM could not fit y ~ gene_fraction:gene + (1 | " f"{gene_column}/{guide_column}) on this frame: " f"{type(error).__name__}: {error}. The nesting needs several " f"genes, several guides inside at least some of them, and more " f"wells than genes. Choose a fixed-effects regression_type with " f"level='gene' or level='grna' if this screen cannot support " f"it.") from error df['residuals'] = mixed_model.resid plot_histogram(df, 'residuals', dst=dst) fixed_names = set(map(str, mixed_model.fe_params.index)) coefs = mixed_model.params p_values = mixed_model.pvalues term_types = [TERM_FIXED if str(name) in fixed_names else TERM_VARIANCE for name in coefs.index] parameter_p = np.asarray(p_values.values, dtype=float) parameter_p = np.where( np.array(term_types) == TERM_VARIANCE, np.nan, parameter_p) frames = [pd.DataFrame({ 'feature': [str(name) for name in coefs.index], 'coefficient': np.asarray(coefs.values, dtype=float), 'p_value': parameter_p, 'term_type': term_types, })] blups = {} for group_key, values in (mixed_model.random_effects or {}).items(): for key, value in dict(values).items(): guide = _blup_guide_name(key) if guide is None: continue blups[guide] = float(value) if blups: guides = sorted(blups) frames.append(pd.DataFrame({ 'feature': [BLUP_FEATURE_TEMPLATE.format(g) for g in guides], 'coefficient': [blups[g] for g in guides], 'p_value': np.full(len(guides), np.nan, dtype=float), 'term_type': [TERM_BLUP] * len(guides), })) coef_df = pd.concat(frames, ignore_index=True) n_blups = int((coef_df['term_type'] == TERM_BLUP).sum()) print(f"Mixed model fitted by regression_backend={backend_label(backend)}") print(f"Mixed model: gene fixed, guide random nested in gene " f"({groups.nunique()} genes, {n_blups} guide BLUPs). " f"A BLUP has no p-value, so results_grna.csv from a mixed run is a " f"shrunken prediction per guide and carries no q value.") if not bool(getattr(mixed_model, 'converged', True)): variances = ', '.join( f"{name}={value:.3g}" for name, value in zip(vc_formula, np.atleast_1d(np.asarray(mixed_model.vcomp, dtype=float)))) print("\n" " ###############################################################\n" " # WARNING: the mixed model did not converge. #\n" " ###############################################################\n" f" Variance components: {variances}; group variance " f"{float(np.asarray(mixed_model.cov_re).ravel()[0]):.3g}.\n" " A variance on the boundary at zero is the usual cause and is\n" " itself an answer, but the standard errors and p-values of the\n" " gene fixed effects are not trustworthy while it stands. Fit a\n" " fixed-effects regression_type with level='gene' to get gene\n" " effects whose intervals can be reported.\n") return mixed_model, coef_df
[docs] def check_and_clean_data(df, dependent_variable): """Prepare the merged count / score frame for model fitting. Drops rows with a missing ``fraction`` or dependent variable, casts the identifier columns to categorical and reports (without dropping) collinear columns via VIF. The returned frame keeps only ``fraction``, the dependent variable, ``gene``, ``grna``, ``prc``, ``plateID``, ``rowID``, ``columnID``, and ``cell_count`` and ``screenID`` when present, plus a computed ``gene_fraction`` column: the sum of the gene's gRNA fractions within each well, which the regression formula regresses on. :param df: Merged DataFrame of counts and scores. :param dependent_variable: Name of the response column. :returns: The cleaned DataFrame used as the model input. :raises ValueError: if a ``(prc, grna)`` pair carries more than one ``fraction``, which makes ``gene_fraction`` ambiguous. """ def handle_missing_values(df, columns): """Handle missing values in specified columns.""" missing_summary = df[columns].isnull().sum() print("Missing values summary:") print(missing_summary) df_cleaned = df.dropna(subset=columns).copy() if df_cleaned.shape[0] < df.shape[0]: print(f"Dropped {df.shape[0] - df_cleaned.shape[0]} rows with missing values in {columns}.") return df_cleaned def ensure_valid_types(df, columns): """Ensure that specified columns are categorical.""" for col in columns: if not isinstance(df[col].dtype, pd.CategoricalDtype): df[col] = pd.Categorical(df[col]) print(f"Converted {col} to categorical type.") return df def check_collinearity(df, columns): """Check for collinearity using VIF (Variance Inflation Factor).""" print("Checking for collinearity...") df_encoded = df[columns] df_encoded = df_encoded.apply(pd.to_numeric, errors='coerce') if np.linalg.matrix_rank(df_encoded.values) < df_encoded.shape[1]: print("Warning: Perfect multicollinearity detected! Dropping correlated columns.") df_encoded = df_encoded.loc[:, ~df_encoded.columns.duplicated()] vif_data = pd.DataFrame() vif_data["Feature"] = df_encoded.columns try: vif_data["VIF"] = [variance_inflation_factor(df_encoded.values, i) for i in range(df_encoded.shape[1])] except np.linalg.LinAlgError: print("LinAlgError: Unable to compute VIF due to matrix singularity.") return df_encoded print("Variance Inflation Factor (VIF) for each feature:") print(vif_data) high_vif_columns = vif_data[vif_data["VIF"] > 10]["Feature"].tolist() if high_vif_columns: print(f"Warning: high collinearity (VIF > 10) for: {high_vif_columns}. " f"Keeping them - the regression formula requires both - but " f"coefficient estimates may be unstable.") return df_encoded df = handle_missing_values(df, ['fraction', dependent_variable]) df = ensure_valid_types(df, ['grna', 'gene', 'plateID', 'rowID', 'columnID', 'prc']) df_cleaned = check_collinearity(df, ['fraction', dependent_variable]) df_cleaned['gene'] = df['gene'] df_cleaned['grna'] = df['grna'] df_cleaned['prc'] = df['prc'] df_cleaned['plateID'] = df['plateID'] df_cleaned['rowID'] = df['rowID'] df_cleaned['columnID'] = df['columnID'] if 'cell_count' in df.columns: df_cleaned['cell_count'] = df['cell_count'] from .schema import SCREEN_KEY if SCREEN_KEY in df.columns: df_cleaned[SCREEN_KEY] = df[SCREEN_KEY] grna_key = ['prc', 'gene', 'grna'] per_grna = df_cleaned[grna_key + ['fraction']].drop_duplicates() clash = per_grna.duplicated(subset=grna_key, keep=False) if clash.any(): offenders = per_grna.loc[clash, grna_key].drop_duplicates() raise ValueError( f"{len(offenders)} (well, gRNA) pair(s) carry more than one " f"'fraction', so the gene's share of the well is ambiguous - e.g. " f"{offenders.iloc[0].to_dict()}. This means the count table was " f"joined twice, or two count files describe the same plate. " f"Aggregate the counts per (prc, grna) before regressing.") gene_totals = per_grna.groupby(['prc', 'gene'], observed=False)['fraction'].sum() df_cleaned['gene_fraction'] = pd.MultiIndex.from_arrays( [df_cleaned['prc'], df_cleaned['gene']]).map(gene_totals) print("Data is ready for model fitting.") return df_cleaned
[docs] def minimum_cell_simulation(settings, num_repeats=10, sample_size=100, tolerance=0.02, smoothing=10, increment=10, dst=None): """ Estimate the minimum number of cells per well needed for a stable well mean. For the wells with the most objects, repeatedly subsamples cells at increasing sample sizes and records the mean absolute difference from the well's full mean. Plots the smoothed curve with a ±1 s.d. band, marks the elbow point (or ``settings['min_cells_per_well']`` when it is set) and writes ``cell_min_threshold.pdf`` into ``dst``. Pass ``dst`` to keep the figure in a specific run folder. When omitted, the function uses the screen-level ``results`` folder derived from ``count_data`` for compatibility with direct notebook and script calls. :param settings: Requires ``score_data`` (CSV path or list of paths), ``dependent_variable``, ``tolerance`` (int percent or float fraction) and ``min_cells_per_well``. ``count_data`` is needed only when ``dst`` is left unset, and only to locate the figure. :param num_repeats: Subsamples drawn per sample size. Default ``10``. :param sample_size: Number of wells, taken largest-first by cell count, to simulate. Default ``100``. :param tolerance: Unused; the tolerance applied is ``settings['tolerance']``. :param smoothing: Rolling-window width used to smooth the curve. :param increment: Step between the simulated sample sizes. :param dst: Folder for ``cell_min_threshold.pdf``, created if missing. Default ``None``: ``<folder of settings['count_data'][0]>/results``. :returns: The elbow point's sample size, i.e. the minimum cell count per well, for passing to :func:`process_scores`. :raises ValueError: if ``settings['tolerance']`` is neither an int nor a float. """ from .utils import correct_metadata_column_names if isinstance(settings['score_data'], str): settings['score_data'] = [settings['score_data']] dfs = [] for i, score_data in enumerate(settings['score_data']): df = tabular.read_table(score_data) df = correct_metadata_column_names(df) df['plateID'] = f'plate{i + 1}' if 'prc' not in df.columns: df['prc'] = _compose_prc_column(df) dfs.append(df) df = pd.concat(dfs, axis=0) cell_counts = df.groupby('prc').size().reset_index(name='cell_count') top_wells = cell_counts.nlargest(sample_size, 'cell_count')['prc'] df = df[df['prc'].isin(top_wells)] diff_data = [] for i, (prc, group) in enumerate(df.groupby('prc')): original_mean = group[settings['dependent_variable']].mean() max_cells = len(group) sample_sizes = np.arange(2, max_cells + 1, increment) for sample_size in sample_sizes: abs_diffs = [] for _ in range(num_repeats): sample = group.sample(n=sample_size, replace=False) sampled_mean = sample[settings['dependent_variable']].mean() abs_diff = abs(sampled_mean - original_mean) abs_diffs.append(abs_diff) avg_abs_diff = np.mean(abs_diffs) diff_data.append((sample_size, avg_abs_diff)) diff_df = pd.DataFrame(diff_data, columns=['sample_size', 'avg_abs_diff']) summary_df = diff_df.groupby('sample_size').agg( mean_abs_diff=('avg_abs_diff', 'mean'), std_abs_diff=('avg_abs_diff', 'std') ).reset_index() summary_df['smoothed_mean_abs_diff'] = summary_df['mean_abs_diff'].rolling(window=smoothing, min_periods=1).mean() if isinstance(settings['tolerance'], int): tolerance_fraction = settings['tolerance'] / 100 elif isinstance(settings['tolerance'], float): tolerance_fraction = settings['tolerance'] else: raise ValueError("Tolerance must be an integer 0 - 100 or float 0.0 - 1.0.") relative_thresholds = { prc: tolerance_fraction * group[settings['dependent_variable']].mean() for prc, group in df.groupby('prc') } summary_df['relative_threshold'] = summary_df['sample_size'].map( lambda size: np.mean([relative_thresholds[prc] for prc in top_wells]) ) elbow_df = summary_df[summary_df['smoothed_mean_abs_diff'] <= summary_df['relative_threshold']] if not elbow_df.empty: elbow_point = elbow_df.iloc[0] else: elbow_point = summary_df.iloc[-1] if dst is None: dst = os.path.join(os.path.dirname(settings['count_data'][0]), 'results') dst = os.path.abspath(os.path.expanduser(os.fspath(dst))) os.makedirs(dst, exist_ok=True) mark = (elbow_point['sample_size'] if settings['min_cells_per_well'] is None else settings['min_cells_per_well']) fig_file_path = _draw_the_cell_count_sweep( summary_df, mark, os.path.join(dst, 'cell_min_threshold.pdf')) if fig_file_path: print(f"Saved {fig_file_path}") return elbow_point['sample_size']
def _statsmodels_p_values(model, coefs): """Return per-coefficient p-values from a statsmodels-shaped results object. Every statsmodels results class spaCR fits exposes ``pvalues``. The fallback exists for :mod:`spacr.power_model`, whose Laplace approximation reports standard errors rather than a test: a two-sided normal p-value from ``coef / bse`` is exactly what a Wald test on that approximation is, and computing it here keeps the horseshoe fit in the same table as the rest instead of giving it a private code path. :param model: Fitted results object. :param coefs: Its ``params``, already extracted. :returns: 1-D float array aligned with ``coefs``. :raises ValueError: when the object carries neither ``pvalues`` nor ``bse``, so no inference is possible. """ pvalues = getattr(model, 'pvalues', None) if pvalues is not None: return np.asarray(pvalues, dtype=float).reshape(-1) bse = getattr(model, 'bse', None) if bse is None: raise ValueError( f"{type(model).__name__} exposes neither .pvalues nor .bse, so " f"spaCR cannot attach a p-value to its coefficients. A results " f"object handed to process_model_coefficients must carry one or " f"the other.") std_err = np.asarray(bse, dtype=float).reshape(-1) with np.errstate(divide='ignore', invalid='ignore'): z = np.where(std_err > 0, np.asarray(coefs, dtype=float).reshape(-1) / std_err, 0.0) return 2.0 * (1.0 - st.norm.cdf(np.abs(z))) def _bootstrap_wald_p_values(model, X, y, n_boot=200, random_state=0): """Return bootstrap Wald p-values for an estimator with no inference. Refits ``model``'s estimator on ``n_boot`` nonparametric resamples of the rows, takes the empirical standard deviation of each coefficient across the resamples and reports ``2 * (1 - Phi(|coef| / sd))``. This is the honest minimum for the hinge backend: an SVM has no likelihood, so there is no Wald or likelihood-ratio test to run, and the alternative - leaving ``p_value`` NaN - would make :func:`perform_regression` select ``p_value <= 0.05`` on an all-NaN column and report "0 significant gRNAs" for every hinge run, which reads exactly like a screen with no hits. A resample that loses a class entirely is skipped rather than fitted; a coefficient whose bootstrap standard deviation is zero (never selected, or identical in every resample) gets ``p = 1``, never a division by zero. :param model: A fitted scikit-learn estimator; cloned, never refitted in place, so the caller's model object is untouched. :param X: Design matrix. :param y: Response the model was fitted on - for hinge, the BINARISED one. :param n_boot: Number of resamples. Default 200. :param random_state: Seed, so a hit list is reproducible from the settings. :returns: 1-D float array of length ``X.shape[1]``. :raises RuntimeError: when no resample could be fitted at all. """ rng = np.random.default_rng(random_state) X_values = np.asarray(X, dtype=float) y_values = np.asarray(y, dtype=float).reshape(-1) n = X_values.shape[0] draws = [] one_class = 0 unfittable = 0 last_failure = None for _ in range(int(n_boot)): idx = rng.integers(0, n, size=n) y_boot = y_values[idx] if np.unique(y_boot).size < 2: one_class += 1 continue try: fitted = clone(model).fit(X_values[idx], y_boot) except Exception as exc: unfittable += 1 last_failure = exc continue draws.append(np.asarray(fitted.coef_, dtype=float).ravel()) if not draws: raise RuntimeError( f"none of the {n_boot} bootstrap resamples could be fitted, so no " f"standard error is available for the hinge coefficients. This " f"usually means one class holds only a handful of wells; check " f"hinge_threshold.") dropped = int(n_boot) - len(draws) if dropped: LOG.warning( "hinge bootstrap: %d of %d resamples produced no coefficients " "(%d were one-class, %d would not fit%s). The p-values below are " "computed from the remaining %d.", dropped, int(n_boot), one_class, unfittable, f"; last error: {last_failure}" if last_failure is not None else "", len(draws)) if len(draws) < 2: LOG.warning( "hinge bootstrap: only %d resample(s) survived, so the coefficient " "standard deviation is zero and EVERY p-value below is exactly " "1.0. That is an absence of evidence, not evidence of absence — " "do not read it as 'no significant gRNAs'.", len(draws)) coefs = np.asarray(model.coef_, dtype=float).ravel() sd = np.std(np.vstack(draws), axis=0, ddof=1) if len(draws) > 1 else \ np.zeros_like(coefs) with np.errstate(divide='ignore', invalid='ignore'): z = np.where(sd > 0, coefs / sd, 0.0) return 2.0 * (1.0 - st.norm.cdf(np.abs(z))) #: Backends whose fitted results object carries ``params`` and ``pvalues`` #: directly. Most are statsmodels; ``horseshoe`` and ``rra`` are spaCR's own #: adapters (:class:`_HorseshoeResults`, :class:`_RRAResults`), which exist so #: that a model with a posterior or a permutation null lands in the same table #: as the likelihood fits instead of getting a private code path. #: ``mixed`` is here too, and its variance components are dropped below - they #: are not effects on the response. _STATSMODELS_COEF_TYPES = ( 'ols', 'wls', 'rlm', 'huber', 'glm', 'poisson', 'logit', 'probit', 'quasi_binomial', 'quantile', 'mixed', 'horseshoe', 'rra', 'spline', ) #: Backends whose fitted object exposes ``coef_`` and ``predict`` and carries #: no inference of its own, so :func:`calculate_p_values` supplies the #: (deliberately conservative) p-value. Three are scikit-learn's; ``group_lasso`` #: is :class:`_GroupLassoResults` around :mod:`spacr.group_lasso`, and it is #: here rather than in a branch of its own precisely so it reports what the #: other penalised backends report. _SKLEARN_COEF_TYPES = ('ridge', 'lasso', 'elasticnet', 'group_lasso') #: The level term patsy writes for one gRNA or one gene: #: ``fraction:grna[224750_2]`` or ``gene_fraction:gene[T.224750]``. Anchored, #: so a nuisance column -- ``Intercept``, ``rowID[T.r2]``, ``columnID[T.c7]``, #: ``screenID[T.b]`` -- does not match and is answered with None. An unanchored #: search would read ``r2`` out of ``rowID[T.r2]`` and hand the row and column #: dummies to the grouping as though they were genes. _LEVEL_TERM_IN_FEATURE = re.compile( r'^(?:fraction:grna|gene_fraction:gene)\[(?:T\.)?(.*)\]$') #: The guide number a gRNA id ends with: ``224750_2`` -> gene ``224750``. _GUIDE_SUFFIX = re.compile(r'_\d+$') def _gene_of_design_column(column): """The gene a design column belongs to, or ``None`` for a nuisance term. THE GROUPING BOTH NEW BACKENDS RUN ON. ``group_lasso`` penalises a gene's guide columns as one block and ``rra`` aggregates their ranks, so both need to know which columns are the same gene's -- and the design matrix is all either of them is given. It is parsed from the column name rather than passed in, because the name is what patsy actually built the column from; a second, separately supplied grouping could disagree with it and would then split a gene silently. ``perform_regression`` reduces ``TGGT1_224750_2`` to ``224750_2`` before the fit (the three-token org/gene/guide split it makes on the merged frame), but a caller that fits ``regression_model`` directly may not have, so the trailing guide number is stripped from whatever is there and the ORG PREFIX IS LEFT ALONE: it is constant across a screen, so ``TGGT1_224750`` and ``224750`` are each a consistent key for their own frame, and stripping it would be a guess about naming. :param column: a design-matrix column name. :returns: the gene id, or ``None`` when the column is not a level term. """ text = str(column) match = _LEVEL_TERM_IN_FEATURE.match(text) if match is None: return None identifier = match.group(1) if text.startswith('gene_fraction:'): return identifier return _GUIDE_SUFFIX.sub('', identifier) or identifier def _level_term_mask(columns): """A boolean mask of the columns that name a gRNA or a gene. ``dtype=bool`` is not incidental: an empty design gives ``np.array([])``, which is float64, and boolean-indexing a coefficient vector with a float array raises ``IndexError`` instead of selecting nothing. """ return np.array([_gene_of_design_column(column) is not None for column in columns], dtype=bool) def _design_column_groups(columns): """One group label per design column, nuisance terms in groups of their own. :mod:`spacr.group_lasso` groups by label, so every column needs one. A nuisance column gets its OWN name as its label, which makes it a singleton group: it is penalised on its own, exactly as ordinary lasso would, and it can never be pulled into a gene's block and dragged to zero with it. """ return [_gene_of_design_column(column) or str(column) for column in columns] def _say_when_a_control_matched_nothing(coef_df, nc, pc, controls) -> None: """Warn, by name, about a control that selected no coefficient. The consequence is named because the number is not obviously missing: the run completes, the volcano draws, and the effect-size cut is simply measured on nothing. """ counts = coef_df['condition'].value_counts() for value, tag, what in ((nc, 'nc', 'negative_control_id'), (pc, 'pc', 'positive_control_id')): if value in (None, '') or int(counts.get(tag, 0)): continue print(f" WARNING: {what}={value!r} matches no coefficient in this " f"screen, so the baseline and the effect-size cut that read it " f"have nothing to measure. Check it against the guide names in " f"the count table -- spaCR reads a bare id as a GENE and one " f"with an underscore as a GUIDE.") if (controls or []) and not int(counts.get('control', 0)): print(f" WARNING: none of the {len(list(controls))} control(s) named " f"matches a coefficient, so there is no control spread to " f"measure an effect-size cut on.")
[docs] def label_control_condition(features, guides, nc=None, pc=None, controls=None, *, strict: bool = False, verbose: bool = False): """Label every coefficient row ``'nc'``, ``'pc'``, ``'control'`` or ``'other'``. The ``condition`` column: what the volcano colours by, what the results panel offers in "colour by", and -- the reason this is a function rather than four lines inside :func:`process_model_coefficients` -- what the EFFECT-SIZE CUT measures its spread on. A coefficient table without it is a table :meth:`spacr.qt.widgets.regression_results.RegressionResultsPanel. set_threshold_method` answers "No control coefficients, so no effect-size cut" for, which is what every guide-permutation run used to get. Precedence is ``nc``, then ``pc``, then the explicit ``controls`` list, so a guide named in two of them is reported once and always the same way. :param features: the model term per row, e.g. ``fraction:grna[000000_1]``. ``nc`` and ``pc`` are matched as SUBSTRINGS of it, which is how a negative control given as a gene id reaches a term named for a guide. :param guides: the guide identifier per row, matched whole against ``controls``. Both sides are compared as text, so a control list that round-tripped through a settings CSV as integers still matches. :param nc: negative-control identifier, or ``None`` for no negative control. :param pc: positive-control identifier, or ``None``. :param controls: non-targeting guide identifiers, or ``None``. ``None`` means "no control list" and labels nothing -- it is the value :func:`perform_regression` documents for a control-free screen, and the inline version this replaced raised ``TypeError`` on it. :param strict: raise :class:`spacr.control_names.ControlNotFound` when a NAMED ``nc`` or ``pc`` matches nothing. Off by default so a call on a partial frame is not an error; the run turns it on, because there a control matching nothing is a number computed against an empty set. :param verbose: print what each control resolved to and how much it matched. :returns: a :class:`pandas.Series` of labels aligned with ``features``. """ features = pd.Series(features).astype(str) guides = pd.Series(guides).astype(str) guides.index = features.index nc_name = '' if nc is None else str(nc) pc_name = '' if pc is None else str(pc) control_names = {str(name) for name in (controls or [])} from .control_names import rows_for labels = pd.Series('other', index=features.index, dtype=object) genes = guides.fillna('').str.split('_').str[0] library = list(guides.astype(str).unique()) if control_names: for name in sorted(control_names): mask, note = rows_for(name, guides, genes, names=library) if verbose and note: print(f" {note}") labels[mask.to_numpy()] = 'control' for name, tag in ((pc_name, 'pc'), (nc_name, 'nc')): if not name: continue mask, note = rows_for(name, guides, genes, names=library, strict=strict, label='positive control' if tag == 'pc' else 'negative control') if verbose and note: print(f" {note}") labels[mask.to_numpy()] = tag return labels
[docs] def process_model_coefficients(model, regression_type, X, y, nc, pc, controls, hinge_threshold=None, hinge_n_boot=200): """Return a DataFrame of model coefficients and p-values, one row per term. Every name in :data:`REGRESSION_TYPES` has a branch here. It is the same table for all of them - ``feature``, ``coefficient``, ``p_value``, ``-log10(p_value)``, ``grna``, ``condition`` - because everything downstream (the volcano plot, the hit table, the metadata merge) reads those columns and nothing else. :param model: The fitted object from :func:`regression_model`. :param regression_type: Which backend produced it. :param X: Design matrix, used for the sklearn feature names and for the p-value approximations that need the data back. :param y: Response, likewise. :param nc: Negative-control identifier, matched against the feature name. :param pc: Positive-control identifier. :param controls: Explicit list of control gRNA identifiers. :param hinge_threshold: The binarisation cut used by the hinge fit; the bootstrap below must reproduce the SAME two classes the fit saw. :param hinge_n_boot: Bootstrap resamples used for the hinge p-values. :returns: Coefficient DataFrame with the row/column nuisance terms removed. :raises ValueError: on an unsupported ``regression_type``. """ if regression_type == 'beta': coefs = model.params std_err = model.bse wald_stats = coefs / std_err p_values = 2 * (1 - st.norm.cdf(np.abs(wald_stats))) coef_df = pd.DataFrame({ 'feature': coefs.index, 'coefficient': coefs.values, 'std_err': std_err.values, 'wald_stat': wald_stats.values, 'p_value': p_values, }) elif regression_type in _STATSMODELS_COEF_TYPES: coefs = model.params p_values = _statsmodels_p_values(model, coefs) coef_df = pd.DataFrame({ 'feature': coefs.index, 'coefficient': coefs.values, 'p_value': np.asarray(p_values, dtype=float), }) if regression_type == 'mixed': coef_df = coef_df[coef_df['feature'].isin( [str(c) for c in X.columns])].reset_index(drop=True) elif regression_type in _SKLEARN_COEF_TYPES: coefs = np.asarray(model.coef_).ravel() p_values = calculate_p_values(X, y, model) coef_df = pd.DataFrame({ 'feature': X.columns, 'coefficient': coefs, 'p_value': p_values, }) elif regression_type == 'hinge': coefs = np.asarray(model.coef_).ravel() p_values = _bootstrap_wald_p_values( model, X, binarise_response(y, hinge_threshold, name='dependent variable'), n_boot=hinge_n_boot) coef_df = pd.DataFrame({ 'feature': X.columns, 'coefficient': coefs, 'p_value': p_values, }) else: raise ValueError(f"Unsupported regression type: {regression_type}") coef_df['-log10(p_value)'] = -np.log10(coef_df['p_value']) coef_df['grna'] = ( coef_df['feature'] .str.extract(r'\[(.*?)\]')[0] .str.replace(r'^T\.', '', regex=True) ) coef_df['condition'] = label_control_condition( coef_df['feature'], coef_df['grna'], nc=nc, pc=pc, controls=controls, verbose=True) _say_when_a_control_matched_nothing(coef_df, nc, pc, controls) nuisance = coef_df['feature'].astype(str).str.match( r'^(?:plateID|rowID|columnID|screenID)\[') return coef_df[~nuisance]
def _draw_the_threshold_sweep(settings, res_folder, *, measured: bool = False) -> None: """Draw the guide-fraction sweep without replacing the threshold in force. :param settings: the regression settings, read for the threshold in force and the count tables the sweep reads. :param res_folder: this run's folder, kept in the signature because the caller collects what the sweep drew into it. :param measured: the threshold in force came from the control-well calibration rather than from the user. THE TWO ANSWERS, SIDE BY SIDE, AND NAMED. The sweep answers "how many guides per well do I want" from the counts alone; the calibration answers "which cut-off makes imaging and sequencing agree" from the control wells. They are different questions, so neither replaces the other and the run reports both -- but a run that says "you set 0.0168" about a number the calibration measured is telling the user something untrue about where their threshold came from, which is the one thing this line exists to say. Plotting is diagnostic; a rendering failure is reported without invalidating the regression run. """ try: chosen = settings.get('fraction_threshold') derived = _graph_sequencing_stats(settings) if derived is not None and chosen is not None: whose, mine = (("the control-well calibration measured", "The measured value") if measured else ("you set", "Your value")) print(f"gRNA fraction-threshold sweep drawn: {whose} " f"{chosen}; the sweep's own pick on this screen is " f"{derived}. The two answer different questions -- which " f"cut-off the control wells agree at, and how many guides " f"a well should keep -- so neither replaces the other. " f"{mine} is the one in force.") except Exception as error: # noqa: BLE001 print(f"the gRNA fraction-threshold sweep could not be drawn " f"({type(error).__name__}: {error}); the run is unaffected " f"and fraction_threshold={settings.get('fraction_threshold')} " f"is still in force") def _show_response_distribution(before_df, dependent_variable, settings): """Display the response distribution before and after transformation. The panel is emitted through Matplotlib so the Qt bridge can add it to the figure queue. It is also shown when no transformation is selected, making the unchanged distribution explicit. """ if before_df is None or not settings.get('plot', True): return try: import matplotlib.pyplot as plt from .response_distribution import panel wanted = [str(dependent_variable), str(settings.get('dependent_variable') or "")] column = next((c for c in wanted if c and c in before_df), None) if column is None: numeric = [c for c in before_df.columns if pd.api.types.is_numeric_dtype(before_df[c])] column = numeric[-1] if numeric else None if column is None: print("the response distribution panel was not drawn: the " "aggregated table carries no numeric response column") return _draw_response_panel_in_pyqtgraph( before_df[column].to_numpy(dtype=float), str(settings.get('transform') or 'none'), str(column), settings.get('src')) except Exception as error: # noqa: BLE001 print(f"the response distribution panel could not be drawn " f"({type(error).__name__}: {error}); the run is unaffected")
[docs] def check_distribution(y, epsilon=1e-6): """Check the distribution of ``y`` and recommend a regression type. :param y: Response vector. :param epsilon: How close to 0 or 1 a value may sit before it counts as a boundary case. Default ``1e-6``. :returns: One of ``'logit'``, ``'quasi_binomial'``, ``'beta'``, ``'ols'`` or ``'glm'``, as accepted by :func:`regression`'s ``regression_type``. """ if np.all((y == 0) | (y == 1)): print("Detected binary data.") return 'logit' elif (y > 0).all() and (y < 1).all(): if np.any((y < epsilon) | (y > 1 - epsilon)): print("Detected continuous data near 0 or 1. Using quasi-binomial.") return 'quasi_binomial' else: print("Detected continuous data between 0 and 1 (no boundary issues). Using beta regression.") return 'beta' elif (y >= 0).all() and (y <= 1).all(): print("Detected continuous data with boundary values (0 or 1). Using quasi-binomial.") return 'quasi_binomial' stat, p_value = stats.normaltest(y) print(f"Normality test p-value: {p_value:.4f}") if p_value > 0.05: print("Detected normally distributed data. Using OLS.") return 'ols' if stats.kstest(y, 'beta', args=(2, 2)).pvalue > 0.05: if np.any((y < epsilon) | (y > 1 - epsilon)): print("Detected continuous data near 0 or 1. Using quasi-binomial.") return 'quasi_binomial' else: print("Detected continuous data between 0 and 1 (no boundary issues). Using beta regression.") return 'beta' print("Detected non-normally distributed data. Using GLM.") return 'glm'
MIN_POISSON_SAMPLES = 8 def _validate_poisson_response(y, X=None, minimum_samples=MIN_POISSON_SAMPLES, model="Poisson regression"): """Validate a response before fitting a Poisson GLM. Poisson endog must contain finite, non-negative integer counts. At least eight observations and one residual degree of freedom are required so family detection and coefficient inference are not performed on an undersized or saturated design. :param y: One-dimensional count response. :param X: Optional design matrix used to determine the parameter count. :param minimum_samples: Absolute observation floor. :param model: What to call the model in the refusal. `horseshoe` is a sparse Poisson GLM and reaches this validator too, so a user who chose it was told "Poisson regression requires integer count data" -- an error naming a model they did not ask for, followed by advice ("use a continuous response model") for a choice they never made. :returns: The validated response as a one-dimensional float array. :raises ValueError: If the response or sample size is invalid. """ try: counts = np.asarray(y, dtype=float).reshape(-1) except (TypeError, ValueError) as exc: raise ValueError( f"{model} requires numeric count data." ) from exc if not np.isfinite(counts).all(): raise ValueError( f"{model} requires finite count data; remove or impute " "NaN and infinite response values before fitting." ) if np.any(counts < 0): raise ValueError( f"{model} requires non-negative count data; negative " "response values are not valid counts." ) if not np.all(np.isclose(counts, np.rint(counts), rtol=0, atol=1e-8)): raise ValueError( f"{model} requires integer count data; use a continuous " "response model for fractional values." ) if not np.any(counts > 0): raise ValueError( f"{model} requires at least one positive count; an " "all-zero response cannot estimate effects." ) n_parameters = 0 if X is not None: x_shape = np.shape(X) if not x_shape or x_shape[0] != counts.size: raise ValueError( f"{model} requires X and y to contain the same " f"number of observations; got {x_shape[0] if x_shape else 0} " f"and {counts.size}." ) n_parameters = 1 if len(x_shape) == 1 else int(x_shape[1]) required = max(int(minimum_samples), n_parameters + 1) if counts.size < required: raise ValueError( f"{model} has too few observations: " f"received {counts.size}, but at least {required} are required " f"for {n_parameters} model parameters." ) return counts #: Transforms that are THEMSELVES a link function. Applying one and then #: handing the result to a family whose link does the same job transforms the #: response twice, and the model fits a quantity nothing measures. LINK_LIKE_TRANSFORMS = ('log', 'logit')
[docs] def double_transform_warning(name, transform, family) -> str: """Describe a response transform that is compounded by the model link. For example, a log-transformed response passed to a family with a logit link fits ``logit(log(y))``. The function returns an actionable warning before fitting; an identity link or a response without a link-like transform returns an empty string. :param name: Response name shown in the warning. :param transform: Transform already applied to the response. :param family: Statsmodels family whose link will be inspected. :returns: Warning text, or ``""`` when the transforms do not compound. """ kind = str(transform or '').strip().lower() if kind not in LINK_LIKE_TRANSFORMS: return "" link = type(getattr(family, 'link', None)).__name__ if link in ('', 'NoneType', 'Identity', 'identity'): return "" return ( f" Warning: {name or 'the response'} would be transformed TWICE. " f"transform={kind!r} has already been applied to it, and the " f"selected family also carries a {link} link, so the model fits " f"{link.lower()}({kind}(y)) -- which is usually why McFadden's " f"R-squared comes back negative and meaningless. " f"spaCR will drop the transform and let the family's link do the " f"work, so it is applied once. To fit the TRANSFORMED response " f"instead, use regression_type='ols' on the transformed response." )
#: A link-like transform and a GLM's own link are the same operation asked #: for twice, and there is one right answer: fit the response AS MEASURED #: and let the family's link do the transforming, once. #: #: This used to be a setting with three values. The other two were not #: choices worth offering: 'transformed' fitted a Gaussian identity model of #: the transformed response, which is an ordinary linear model and exactly #: what regression_type='ols' already gives; and 'warn' kept the double #: transform so an older result could be reproduced. A user who wants the #: first still has it under its own name, and the second was a bug being #: preserved.
[docs] def resolve_glm_transform_conflict(dependent_variable, transform='', available=(), regression_type='glm'): """Resolve a transform/family-link conflict before fitting a GLM. :param dependent_variable: the response column as it stands -- already the transformed one, if a transform was asked for. :param transform: the transform already applied. :param available: the column names the frame actually holds. Used to confirm the untransformed column is there before switching to it. :param regression_type: only ``'glm'`` chooses its own family, so only ``'glm'`` has this conflict to resolve. Everything else is returned unchanged. :returns: ``(column, transform_in_effect, force_identity, note)``. ``note`` explains any scale change for the run log. A transform that is not link-like, or a regression type other than ``'glm'``, returns the response unchanged. Resolving the column before the design matrices are built keeps the fit, coefficients, diagnostics, and goodness-of-fit summary on the same scale. """ kind = str(transform or '').strip().lower() column = str(dependent_variable) if kind not in LINK_LIKE_TRANSFORMS or str(regression_type) != 'glm': return column, transform, False, '' prefix = f"{kind}_" raw = column[len(prefix):] if column.startswith(prefix) else '' if not raw or raw not in set(available): return column, transform, False, ( f" the response before transform={kind!r} is not in the frame " f"(looked for {raw or 'a column without the prefix'}), so " f"{column} is fitted as it stands and the transform is applied " f"twice -- once by hand and once by the family's link.") return raw, '', False, ( f" fitting the measured response {raw} and ignoring " f"transform={kind!r}: the family's own link does the transforming, " f"so it is applied once instead of twice.")
def _choose_glm_family(y, name="", transform=""): """The family and link, and the sentence saying which scale was examined.""" values = np.asarray(y, dtype=float).reshape(-1) scale = str(name or 'the response') if transform: scale = f"{scale} (after transform={str(transform)!r})" if np.all((values == 0) | (values == 1)): print(f"{scale} is binary. Using Binomial family with Logit link.") return sm.families.Binomial(link=sm.families.links.Logit()) elif (values > 0).all() and (values < 1).all(): print(f"{scale} is strictly between 0 and 1. Using Binomial family " f"with Logit link; consider regression_type='beta', which models " f"the variance of a bounded response directly, or " f"'quasi_binomial' if the wells are overdispersed.") return sm.families.Binomial(link=sm.families.links.Logit()) elif (values >= 0).all() and (values <= 1).all(): print(f"{scale} is between 0 and 1 including the boundaries. " f"Using Quasi-Binomial.") return sm.families.Binomial(link=sm.families.links.Logit()) if (values >= 0).all() and np.all(values.astype(int) == values): _validate_poisson_response(values, minimum_samples=1) print(f"{scale} looks like counts. Using Poisson with Log link.") return sm.families.Poisson(link=sm.families.links.Log()) stat, p_value = normaltest(values) print(f"Normality test p-value: {p_value:.4f}") if p_value > 0.05: print(f"{scale} is normally distributed. Using Gaussian with " f"Identity link.") return sm.families.Gaussian(link=sm.families.links.Identity()) if ((values > 0).all() and kstest(values, 'invgauss', args=(1,)).pvalue > 0.05): print(f"{scale} looks inverse Gaussian. Using InverseGaussian " f"with Log link.") return sm.families.InverseGaussian(link=sm.families.links.Log()) if (values >= 0).all(): print(f"{scale} looks like overdispersed counts. Using Negative " f"Binomial with Log link.") return sm.families.NegativeBinomial(link=sm.families.links.Log()) print(f"{scale}: no family fitted the shape, so Gaussian with an " f"Identity link is used.") return sm.families.Gaussian(link=sm.families.links.Identity())
[docs] def binarise_response(y, threshold=None, name='response'): """Return ``y`` as a 0/1 vector for a classifier backend, refusing to guess. The hinge backend fits a decision boundary, so it needs two classes. There are exactly two ways to get them and this function will not invent a third: * ``y`` already holds exactly two distinct finite values (the usual case: a per-object class call aggregated to a well, or a 0/1 score). The lower value becomes 0 and the higher becomes 1, so the sign of every coefficient answers "does this gRNA push wells towards the HIGHER class", which is the same direction the continuous models report. * ``threshold`` is given explicitly, and ``y > threshold`` becomes 1. A continuous response with no threshold is REFUSED. Picking a cut for the user — the mean, the median, 0.5 — would silently redefine the hypothesis being tested: on a screen whose well scores run 0.2-0.8 a median split calls half the plate positive by construction, and the resulting hit list is a plausible, unfalsifiable artefact of the split. :param y: Response vector (array, Series or single-column frame). :param threshold: Explicit cut; values strictly greater become 1. :param name: Name used in error messages, for a legible failure. :returns: ``numpy`` float array of 0.0/1.0, same length as ``y``. :raises ValueError: if ``y`` is continuous and no ``threshold`` is given, if a given ``threshold`` puts every observation in one class, or if ``y`` holds fewer than two distinct values. Example: .. code-block:: python binarise_response([0, 1, 1, 0]) # -> [0., 1., 1., 0.] binarise_response([2, 5, 5], ) # -> [0., 1., 1.] binarise_response([0.2, 0.6], threshold=0.4) # -> [0., 1.] """ values = np.asarray(y, dtype=float).reshape(-1) if not np.isfinite(values).all(): raise ValueError( f"hinge regression requires a finite {name}; remove or impute the " f"NaN/infinite values before fitting.") if _left_blank(threshold): threshold = None if threshold is not None: cut = float(threshold) binary = (values > cut).astype(float) n_positive = int(binary.sum()) if n_positive == 0 or n_positive == binary.size: raise ValueError( f"hinge_threshold={cut!r} puts all {binary.size} observations " f"in one class ({name} range " f"{values.min():.6g}-{values.max():.6g}); a one-class response " f"has no decision boundary to fit.") return binary unique = np.unique(values) if unique.size == 2: return (values == unique[1]).astype(float) if unique.size < 2: raise ValueError( f"hinge regression needs two classes but {name} holds the single " f"value {unique[0]!r}.") raise ValueError( f"hinge regression needs a binary {name}, but it holds " f"{unique.size} distinct values in " f"{values.min():.6g}-{values.max():.6g}. Set hinge_threshold to the " f"cut you mean (values strictly above it are the positive class), or " f"choose a model for a continuous response ('ols', 'beta', " f"'quantile'). spaCR will not pick the cut for you: a split chosen by " f"the software decides the hypothesis, not the biology.")
def _left_blank(value) -> bool: """Whether a policed setting was left empty rather than answered. None is the usual empty; ``''`` is what a Qt line edit and a saved settings CSV produce for the same untouched box; whitespace is what a hand-edited CSV produces. None of the policed settings takes a string value, so a blank one can only mean "not answered". AND NaN, which is the FOURTH spelling of empty and the one that actually reaches this function. `pandas.read_csv` turns an empty cell into `float('nan')`, so a settings CSV with `hinge_threshold,` on a line -- which is what a saved file looks like for every box the user did not fill -- arrived here as a float. It is not None and not a str, so it read as "answered", and an ordinary OLS run was refused with regression_type='ols' does not read hinge_threshold=nan about a value nobody typed. NaN is never a threshold, a covariance type or a quantile, so there is no reading of it that means "answered". """ if value is None: return True if isinstance(value, str): return not value.strip() try: return bool(value != value) except Exception: # noqa: BLE001 return False def _reject_unused_settings(regression_type, supplied): """Raise when a setting the chosen backend cannot read was set anyway. ``supplied`` maps a setting name to ``(value, default)``. A value equal to its default is "not asked for" and passes; anything else must appear in :data:`REGRESSION_SETTINGS_USED` for this type. Comparing against the default is what makes this usable from a GUI, which posts every widget on the panel whether or not the user touched it. A BLANK IS NOT A REQUEST. An empty box in the panel, and the empty cell a saved settings CSV writes for it, both arrive here as ``''`` -- which is not equal to a default of ``None`` and was therefore refused. The symptom was that the screen's OWN saved settings could not be reloaded and refitted under a different regression type: `hinge_threshold` had never been typed into, and switching to 'ols' raised on it. :param regression_type: The backend about to be fitted. :param supplied: ``{name: (value, default)}`` for the policed settings. :raises ValueError: naming the setting, the type and the alternative. """ used = REGRESSION_SETTINGS_USED.get(regression_type, ()) for name, (value, default) in supplied.items(): if name in used or value == default or _left_blank(value): continue raise ValueError( f"regression_type={regression_type!r} does not read {name}=" f"{value!r}: {_SETTING_NOT_APPLICABLE[name]} Leave {name} at its " f"default ({default!r}), or choose a regression type that uses it " f"({', '.join(t for t in REGRESSION_TYPES if name in REGRESSION_SETTINGS_USED[t]) or 'none'}).") #: Why each policed setting does nothing for the types that do not list it. #: Split out of :func:`_reject_unused_settings` so the message names the #: actual reason instead of "not supported". _SETTING_NOT_APPLICABLE = { 'alpha': "it is the penalty weight of a penalised fit, and this model is " "unpenalised, so the number would change nothing.", 'l1_ratio': "it splits a penalty between L1 and L2, and only 'elasticnet' " "has both.", 'cov_type': "it selects a sandwich covariance estimator on a likelihood " "fit; sklearn's penalised estimators and the robust/quantile " "fits do not expose one, so the standard errors would come " "from somewhere other than the label suggests.", 'quantile': "it is the quantile of the conditional distribution being " "fitted, which only 'quantile' regression has; every other " "model fits the mean (or the median, for 'rlm').", 'hinge_threshold': "it is the cut that turns a continuous response into " "the two classes a hinge loss separates; no other " "model classifies.", 'spline_knots': "it sets how many knots each CONTINUOUS covariate's " "basis gets, and only the spline fit builds one; the " "guide columns are untouched either way.", 'spline_degree': "it sets the polynomial degree of that basis, and only " "the spline fit builds one.", 'huber_t': "it is the residual, in units of the estimated scale, at which " "Huber's loss switches from squared to linear; only the robust " "fits have that switch.", 'lasso_n_boot': "it sizes the bootstrap that ranks features by SELECTION " "frequency, and only a penalty that sets coefficients to " "exactly zero selects anything - ridge keeps every " "feature, so its selection frequency is 1.0 by " "construction, and the likelihood fits are ranked by their " "own p-values.", 'lasso_selection_threshold': "it is the cut on that same selection " "frequency, which only the sparse penalties " "produce.", 'hinge_n_boot': "it sizes the bootstrap that stands in for the standard " "errors an SVM does not have; every other model reports " "its own inference.", 'group_lasso_lambda': "it is the block penalty of the group lasso, " "measured against THIS design's own " "group_lasso.max_lambda, so it is not the same " "quantity as 'alpha' and no other model has a " "block to penalise.", 'rra_alpha': "it is the top fraction of the guide ranking alpha-RRA " "aggregates over, and only 'rra' ranks anything; every other " "model estimates coefficients jointly.", 'rra_permutations': "it sizes the permutation null RRA's P value is read " "off, and every other model gets its P value from a " "likelihood, a posterior or a bootstrap.", } #: The design factors :mod:`pyfixest` absorbs instead of carrying as columns. #: #: `prepare_formula` puts ``rowID`` and ``columnID`` in the model as FIXED #: effects, so patsy dummy-codes them. They are nuisance terms -- #: `process_model_coefficients` drops every one of them from the coefficient #: table before anybody reads it -- and a nuisance term that is never reported #: does not have to be a column. Absorbing it by alternating projections #: (Frisch-Waugh-Lovell) leaves the coefficients that ARE reported unchanged #: to the last digit and takes the solve down with the design. #: #: ``screenID`` is deliberately excluded because it blocks combined-screen #: fits on the experiment and may be useful in the coefficient table. _ABSORBED_FIXED_EFFECTS = ('rowID', 'columnID') def _absorbed_factor_codes(X, factors=_ABSORBED_FIXED_EFFECTS): """Recover the level of each factor patsy dummy-coded, per observation. patsy writes a k-level factor as k-1 indicator columns against a dropped reference level, so the reference is the row where every one of them is zero. Reading the codes back out of the design is what lets an absorbing backend be handed the SAME matrix statsmodels was, rather than the raw frame -- there is then no second construction of the design to disagree with the first. :param X: the design DataFrame patsy built. :param factors: term names to look for, each dummy-coded as ``name[T.level]``. :returns: ``(codes, names, n_absorbed_params)`` -- an ``(n, k)`` uint64 array of level codes, the factors actually found, and how many parameters of the dense design they account for (the intercept plus each factor's k-1 indicators, which is what the residual degrees of freedom must still be charged for). ``codes`` is ``None`` when no factor is present. :raises ValueError: when a factor's indicator columns are not 0/1, which means the column named ``rowID[...]`` is not a dummy and absorbing it would silently fit a different model. """ columns = list(getattr(X, 'columns', [])) blocks, names = [], [] n_params = 1 if 'Intercept' in columns else 0 for factor in factors: prefix = f'{factor}[' block = [c for c in columns if str(c).startswith(prefix)] if not block: continue values = np.asarray(X[block], dtype=float) if not np.all(np.isin(values, (0.0, 1.0))): raise ValueError( f"the design's {factor!r} columns are not 0/1 indicators, so " f"they cannot be a dummy-coded factor and absorbing them " f"would fit a different model. Columns: {block[:4]}.") if np.any(values.sum(axis=1) > 1): raise ValueError( f"a row of the design is in more than one {factor!r} level, " f"so {factor!r} is not a factor and cannot be absorbed.") blocks.append(np.where(values.any(axis=1), values.argmax(axis=1) + 1, 0).astype(np.uint64)) names.append(factor) n_params += len(block) if not blocks: return None, [], n_params return np.column_stack(blocks), names, n_params class _AbsorbedDesign: """The design an absorbed fit was run on, in statsmodels' shape. :mod:`spacr.regression_qc` recovers a design from ``results.model.exog`` and decides the scale rule from ``type(results.model).__mro__`` (see ``regression_qc._model_kind``), so an absorbed fit that carried neither would lose the diagnostics tab. It carries the FULL design -- the one with the dummy columns still in it -- because that is the model that was fitted; absorption is how it was solved, not what it was. """ def __init__(self, endog, exog, exog_names): """Store response, full design, and statsmodels-compatible names. :param endog: the response, flattened to one dimension. :param exog: the FULL design, dummy columns included. Not the absorbed one: :mod:`spacr.regression_qc` recovers the design from here, and absorption is how the fit was solved rather than what was fitted. :param exog_names: one name per column of ``exog``, in order. statsmodels' diagnostics index the design by name, so a list that is shorter than the design silently mislabels the columns after the gap rather than raising. """ self.endog = np.asarray(endog, dtype=float).reshape(-1) self.exog = np.asarray(exog, dtype=float) self.exog_names = list(exog_names) #: ``kind`` -> the design class :mod:`spacr.regression_qc` resolves BY NAME. #: #: ``regression_qc._model_kind`` walks ``type(results.model).__mro__`` looking #: for a class called ``OLS`` or ``WLS``, because ``sm.OLS(...).fit()`` and #: ``sm.WLS(...).fit()`` share one results class and only ``results.model`` #: tells them apart. An absorbed least-squares fit obeys exactly the scale #: rule that name selects -- ``scale`` is RSS / (n - p) over the FULL #: parameter count, and the weighted version is in the metric of #: ``sqrt(w) * (y - fitted)`` -- so it answers to it. Built with ``type()`` #: rather than written as two ``class`` statements so ``spacr.ml`` does not #: grow public names ``OLS`` and ``WLS`` that would read as statsmodels'. _ABSORBED_DESIGN_CLASSES = { name: type(name, (_AbsorbedDesign,), {'__doc__': f"The design of an absorbed {name} fit."}) for name in ('OLS', 'WLS') } class _AbsorbedLeastSquaresResults: """A least-squares fit solved by absorbing its nuisance factors. Reports what :func:`process_model_coefficients` and :mod:`spacr.regression_qc` read off a statsmodels results object -- ``params``, ``bse``, ``pvalues``, ``tvalues``, ``resid``, ``fittedvalues``, ``scale``, ``df_resid`` -- for the coefficients that SURVIVE absorption. The absorbed ones have no row, which is the one way this fit's answer differs from statsmodels' and is why ``REGRESSION_BACKENDS['pyfixest']['differs']`` says so. The residuals and ``scale`` are the FULL model's, not the demeaned regression's: Frisch-Waugh-Lovell makes them the same vector, and the degrees of freedom are charged for every absorbed parameter, so the standard errors match statsmodels to the last digit rather than to a tolerance. """ def __init__(self, params, bse, pvalues, resid, fitted, scale, df_model, df_resid, nobs, model, converged, absorbed, rsquared): """Store absorbed-fit estimates and derive their t statistics. :param params: coefficients that SURVIVED absorption, indexed by column name. The absorbed factors have no row here at all, which is the one way this fit's answer differs from statsmodels'. :param bse: their standard errors. :param pvalues: two-sided p values for those coefficients. :param resid: residuals of the FULL model, not of the demeaned regression. Frisch-Waugh-Lovell makes them the same vector. :param fitted: fitted values of the full model. :param scale: residual variance, charged for every absorbed parameter, which is what makes the standard errors match statsmodels to the last digit rather than to a tolerance. :param df_model: model degrees of freedom, full parameters less the intercept -- including the absorbed ones. :param df_resid: residual degrees of freedom after that charge. :param nobs: number of observations. :param model: the design object carrying the surviving column names. :param converged: whether the solver reported convergence. :param absorbed: the factor names that were absorbed and therefore have no coefficient. :meth:`predict` refuses a new row by naming them, because their levels were never estimated. :param rsquared: R-squared of the full model. """ self.params = params self.bse = bse self.pvalues = pvalues with np.errstate(divide='ignore', invalid='ignore'): self.tvalues = params / bse self.resid = resid self.fittedvalues = fitted self.scale = float(scale) self.df_model = float(df_model) self.df_resid = float(df_resid) self.nobs = float(nobs) self.model = model self.converged = bool(converged) self.absorbed = tuple(absorbed) self.rsquared = float(rsquared) def predict(self, exog=None): """Fitted values. ``exog`` is accepted and ignored, as sm's OLS does for the in-sample case; an absorbed fit cannot predict a new row because it never estimated the absorbed levels.""" if exog is None: return self.fittedvalues raise ValueError( "an absorbed fit did not estimate the levels of " f"{', '.join(self.absorbed) or 'its nuisance factors'}, so it " "cannot predict a row it has not seen. Fit with " "regression_backend='statsmodels' if you need out-of-sample " "predictions.") def summary(self): """A text summary, so :func:`_write_model_summary` still has one.""" return _AbsorbedSummary(self) class _AbsorbedSummary: """``.as_text()`` for :class:`_AbsorbedLeastSquaresResults`.""" def __init__(self, results): """Bind the absorbed-fit results rendered by this summary. :param results: the :class:`_AbsorbedLeastSquaresResults` to render. Held, not copied -- the summary is built on demand by :meth:`_AbsorbedLeastSquaresResults.summary` and read once. """ self._results = results def as_text(self): """Return a multiline report of the absorbed least-squares fit.""" r = self._results lines = [ "Absorbed least squares (pyfixest alternating projections)", f" observations {int(r.nobs)}", f" reported coefficients {len(r.params)}", f" absorbed factors {', '.join(r.absorbed) or 'none'}", f" residual df {int(r.df_resid)}", f" error variance {r.scale:.6g}", f" R-squared {r.rsquared:.6f}", f" demeaning converged {r.converged}", "", "coefficient / std err / t / P>|t|", ] for name in r.params.index: lines.append(f" {name} {r.params[name]:.6g} " f"{r.bse[name]:.6g} {r.tvalues[name]:.4f} " f"{r.pvalues[name]:.4g}") return "\n".join(lines) def __str__(self): """Return the same report as :meth:`as_text`.""" return self.as_text() def _fit_absorbed_least_squares(X, y, weights=None, kind='OLS'): """Least squares with ``rowID``/``columnID`` absorbed, via pyfixest. Use ``pyfixest.core.demean`` to project out row and column factors, then solve the remaining normal equations by Cholesky decomposition. This avoids including high-cardinality nuisance dummies in the dense solve while retaining a coefficient table compatible with the statsmodels path. :param X: the design DataFrame, dummy columns included. :param y: the response. :param weights: per-observation weights for a WLS fit, or ``None``. :param kind: ``'OLS'`` or ``'WLS'``, which is what :mod:`spacr.regression_qc` reads to pick its scale rule. :returns: :class:`_AbsorbedLeastSquaresResults`. :raises ValueError: when the design carries no absorbable factor (there is then nothing for this backend to do that statsmodels does not do better), or when the normal equations are singular. """ columns = list(getattr(X, 'columns', [])) if not columns: raise ValueError( "the absorbing backend reads which columns are rowID/columnID " "dummies from the design's COLUMN NAMES, so it needs a DataFrame " "design; a bare array has no names. Build it with " "dmatrices(..., return_type='dataframe'), which is what the " "pipeline hands in.") codes, absorbed, n_absorbed_params = _absorbed_factor_codes(X) if codes is None: raise ValueError( "regression_backend='pyfixest' absorbs the rowID and columnID " "fixed effects, and this design has neither -- either " "model_plate_position=False took them out of the model or the " "screen sits on one row and one column. There is nothing to " "absorb, so the fit would be the statsmodels fit with an extra " "projection in front of it. Set " "regression_backend='statsmodels'.") keep = [c for c in columns if str(c) != 'Intercept' and not any(str(c).startswith(f'{f}[') for f in absorbed)] if not keep: raise ValueError( "every column of this design is an intercept or an absorbed " "fixed effect, so the absorbed fit would report no coefficient " "at all.") y_flat = np.asarray(y, dtype=float).reshape(-1) n = y_flat.size if weights is None: w = np.ones(n, dtype=float) else: w = np.asarray(weights, dtype=float).reshape(-1) if w.size != n: raise ValueError( f"the absorbed fit was given {w.size} weights for {n} " f"observations.") if not np.isfinite(w).all() or np.any(w <= 0): raise ValueError( "WLS weights must be finite and positive (they are per-well " f"cell counts); got {np.nanmin(w)}-{np.nanmax(w)}.") p_full = len(keep) + n_absorbed_params df_resid = n - p_full if df_resid <= 0: raise ValueError( f"the design has {p_full} parameters ({len(keep)} reported plus " f"{n_absorbed_params} absorbed) for {n} observations, so there " f"are no residual degrees of freedom to estimate a standard " f"error from.") from pyfixest.core.demean import demean stacked = np.asfortranarray( np.column_stack([y_flat, np.asarray(X[keep], dtype=float)])) demeaned, converged = demean(stacked, codes, w, tol=1e-10) if not converged: raise ValueError( "the alternating projections that absorb " f"{', '.join(absorbed)} did not converge, so the design was " "never fully partialled out and the coefficients would not be " "the least-squares ones. Fit with " "regression_backend='statsmodels'.") y_d = demeaned[:, 0] X_d = demeaned[:, 1:] Xw = X_d * w[:, None] xtx = X_d.T @ Xw xty = Xw.T @ y_d _rank = int(np.linalg.matrix_rank(xtx)) if _rank < xtx.shape[0]: raise ValueError( f"the absorbed design's normal equations are singular " f"(rank {_rank} of {xtx.shape[0]}), so its {len(keep)} " f"coefficients are not identified. That is a rank-deficient " f"design, not a backend failure: statsmodels answers the same " f"design with a pseudo-inverse, which picks one arbitrary " f"solution out of infinitely many.") beta = np.linalg.solve(xtx, xty) resid = y_d - X_d @ beta rss = float(resid @ (resid * w)) scale = rss / df_resid cov = scale * np.linalg.inv(xtx) se = np.sqrt(np.clip(np.diag(cov), 0.0, None)) with np.errstate(divide='ignore', invalid='ignore'): t_stats = np.where(se > 0, beta / se, 0.0) p_values = 2.0 * st.t.sf(np.abs(t_stats), df_resid) names = [str(c) for c in keep] params = pd.Series(beta, index=names) full_resid = resid fitted = y_flat - full_resid centred = y_flat - np.average(y_flat, weights=w) tss = float(centred @ (centred * w)) rsquared = 1.0 - rss / tss if tss > 0 else float('nan') model = _ABSORBED_DESIGN_CLASSES[kind]( y_flat, np.asarray(X, dtype=float), [str(c) for c in columns]) return _AbsorbedLeastSquaresResults( params=params, bse=pd.Series(se, index=names), pvalues=pd.Series(p_values, index=names), resid=full_resid, fitted=fitted, scale=scale, df_model=p_full - 1, df_resid=df_resid, nobs=n, model=model, converged=converged, absorbed=absorbed, rsquared=rsquared) #: ``regression_type`` -> the glum family, and how the fit is set up. #: #: ``probit`` IS NOT HERE, and that is measured rather than an omission: glum #: 3.4 ships ``IdentityLink``, ``LogLink``, ``LogitLink``, ``CloglogLink`` and #: ``TweedieLink`` and has no probit link at all, so a probit fitted "by glum" #: could only be a logit under the wrong label. ``quasi_binomial`` is not here #: either -- statsmodels spells it as a Binomial mean with the dispersion #: taken from the Pearson chi-square (``scale='X2'``), and glum has no #: equivalent knob, so its standard errors would be the fixed-dispersion ones #: on a model chosen BECAUSE its dispersion is free. `backend_status` greys #: the pair out for exactly these reasons. _GLUM_FAMILIES = { 'poisson': 'poisson', 'logit': 'binomial', 'glm': None, } class _GlumResults: """A GLM fitted by glum, reporting what statsmodels' GLM results report. Form covariance from the canonical-link information matrix, ``(X' W X)^-1`` with ``W_ii = v_i (dmu/deta)^2 / V(mu_i)`` and dispersion fixed at one. This matches the covariance convention used by the statsmodels GLM path rather than glum's optional sandwich or finite-sample corrections. """ def __init__(self, params, bse, pvalues, resid, fitted, scale, df_model, df_resid, nobs, model, family, llf, null_deviance, deviance, n_iter, llnull=None): """Store glum estimates in the statsmodels-compatible results shape. Every argument is named because this class exists to be READ LIKE A statsmodels RESULT, and a reader who cannot tell which of sixteen positional values is the null deviance cannot check that claim. :param params: fitted coefficients, a Series indexed by column name. :param bse: their standard errors, from the canonical-link information matrix described in the class docstring. :param pvalues: two-sided p values for the coefficients. :param resid: response residuals, ``y - mu``. :param fitted: fitted values on the RESPONSE scale, ``mu``. :param scale: the dispersion. One for the fixed-dispersion families and the Pearson estimate for Gaussian, which is the convention :mod:`spacr.regression_qc` reads off ``model.scale``. :param df_model: model degrees of freedom, coefficients less the intercept. :param df_resid: residual degrees of freedom, rows less coefficients. :param nobs: number of observations the fit used. :param model: the design object, carrying the column names and presented as a ``GLM`` class so the QC path resolves it. :param family: the glum family, which supplies ``loglike`` and ``deviance``. :param llf: log-likelihood of the fitted model. :param null_deviance: deviance of the intercept-only model. :param deviance: deviance of the fitted model. :param n_iter: iterations the solver took, 0 when it does not report. :param llnull: log-likelihood of the NULL model, optional. CARRIED RATHER THAN DERIVED. ``fit_quality_note`` falls back to ``null_deviance / -2`` when it is absent, so a backend passing only the deviance prints a different McFadden from statsmodels for the identical fit -- the one thing this class exists not to do. The null model is fitted anyway; this only keeps its answer. """ self.params = params self.bse = bse self.pvalues = pvalues with np.errstate(divide='ignore', invalid='ignore'): self.tvalues = params / bse self.resid = resid self.resid_response = resid self.fittedvalues = fitted self.scale = float(scale) self.df_model = float(df_model) self.df_resid = float(df_resid) self.nobs = float(nobs) self.model = model self.family = family self.llf = float(llf) self.null_deviance = float(null_deviance) self.llnull = None if llnull is None else float(llnull) self.deviance = float(deviance) self.n_iter = int(n_iter) def predict(self, exog=None): """In-sample fitted values on the RESPONSE scale, as sm's GLM does.""" if exog is None: return self.fittedvalues raise ValueError( "this GLM was fitted by glum through spaCR's design matrix and " "does not carry the link's inverse for a new row. Fit with " "regression_backend='statsmodels' if you need to predict.") def summary(self): """Return the text-summary adapter for this fit.""" return _GlumSummary(self) class _GlumSummary: """``.as_text()`` for :class:`_GlumResults`.""" def __init__(self, results): """Bind the glum results rendered by this summary. :param results: the :class:`_GlumResults` to render. Held, not copied; see :class:`_AbsorbedSummary`. """ self._results = results def as_text(self): """Return a multiline report of the glum fit.""" r = self._results lines = [ f"Generalized linear model fitted by glum " f"({type(r.family).__name__})", f" observations {int(r.nobs)}", f" coefficients {len(r.params)}", f" residual df {int(r.df_resid)}", f" deviance {r.deviance:.6g}", f" null deviance {r.null_deviance:.6g}", f" log-likelihood {r.llf:.6g}", f" IRLS steps {r.n_iter}", "", "coefficient / std err / z / P>|z|", ] for name in r.params.index: lines.append(f" {name} {r.params[name]:.6g} " f"{r.bse[name]:.6g} {r.tvalues[name]:.4f} " f"{r.pvalues[name]:.4g}") return "\n".join(lines) def __str__(self): """Return the same report as :meth:`as_text`.""" return self.as_text() def _glum_information_weights(family, mu, var_weights): """``W_ii`` of the GLM information matrix, for the families glum fits. For a canonical link ``dmu/deta`` equals the variance function, so the weight collapses to ``v_i V(mu_i)``: ``v * mu`` for a log-link Poisson and ``v * mu (1 - mu)`` for a logit Binomial. Gaussian identity is ``v``. They are written out per family rather than differenced numerically because a finite difference here would put its own error into every standard error on the volcano. """ weights = np.asarray(var_weights, dtype=float).reshape(-1) mu = np.asarray(mu, dtype=float).reshape(-1) if isinstance(family, sm.families.Poisson): return weights * mu if isinstance(family, sm.families.Binomial): return weights * mu * (1.0 - mu) if isinstance(family, sm.families.Gaussian): return weights raise ValueError( f"spaCR does not know the information weight for " f"{type(family).__name__}, so it cannot form the standard errors of a " f"glum fit of it. Fit with regression_backend='statsmodels'.") def _fit_glum_glm(X, y, regression_type, weights=None, exposure=None): """Fit one of the GLM families through glum instead of statsmodels. Use glum's IRLS and active-set solver for supported Poisson, binomial, or automatically selected GLM families. Small designs may not amortize the backend's setup cost; its advantage is intended for wide model matrices. :param X: the design DataFrame. :param y: the response. :param regression_type: one of :data:`_GLUM_FAMILIES`. :param weights: per-well cell counts, used as ``var_weights`` by the binomial families exactly as the statsmodels path uses them. :param exposure: per-well cell counts for the Poisson ``offset(log(.))``. :returns: :class:`_GlumResults`. :raises ValueError: for a family glum cannot fit, or a response the family refuses. """ columns = list(getattr(X, 'columns', [])) if not columns: raise ValueError( "the glum backend reports one coefficient per design column and " "reads the names from the design, so it needs a DataFrame; a " "bare array has no names.") design = np.asarray(X, dtype=float) y_flat = np.asarray(y, dtype=float).reshape(-1) n = y_flat.size offset = None var_weights = np.ones(n, dtype=float) if regression_type == 'poisson': _validate_poisson_response(y, X) family = sm.families.Poisson(link=sm.families.links.Log()) elif regression_type == 'logit': family = sm.families.Binomial(link=sm.families.links.Logit()) else: family = pick_glm_family_and_link(y) if isinstance(family, sm.families.Poisson): _validate_poisson_response(y, X) if isinstance(family, sm.families.Poisson): n_total = None if exposure is not None: n_total = np.asarray(exposure, dtype=float).reshape(-1) if n_total.size != n: raise ValueError( f"the Poisson exposure has {n_total.size} entries but " f"the response has {n}; each well must carry its own " f"cell count.") if not np.isfinite(n_total).all() or np.any(n_total <= 0): raise ValueError( "the Poisson exposure is the well's cell count, so it " f"must be finite and strictly positive; got " f"{np.nanmin(n_total)}-{np.nanmax(n_total)}.") offset = np.log(n_total) else: print("Warning: no per-well cell count reached the Poisson fit, " "so it models the raw count with no offset(log(cell_count)).") elif isinstance(family, sm.families.Binomial) and weights is not None: var_weights = np.asarray(weights, dtype=float).reshape(-1) if var_weights.size != n: raise ValueError( f"the binomial fit was given {var_weights.size} weights for " f"{n} observations.") glum_family = { 'Poisson': 'poisson', 'Binomial': 'binomial', 'Gaussian': 'normal', }.get(type(family).__name__) if glum_family is None: raise ValueError( f"regression_backend='glum' cannot fit a " f"{type(family).__name__} family; spaCR routes poisson, binomial " f"and gaussian through it. Fit with " f"regression_backend='statsmodels'.") from glum import GeneralizedLinearRegressor estimator = GeneralizedLinearRegressor( family=glum_family, alpha=0, fit_intercept=False, gradient_tol=1e-10, max_iter=500) fit_kwargs = {} if offset is not None: fit_kwargs['offset'] = offset if weights is not None and isinstance(family, sm.families.Binomial): fit_kwargs['sample_weight'] = var_weights estimator.fit(design, y_flat, **fit_kwargs) beta = np.asarray(estimator.coef_, dtype=float).reshape(-1) eta = design @ beta + (0.0 if offset is None else offset) mu = family.link.inverse(eta) info_w = _glum_information_weights(family, mu, var_weights) xtwx = design.T @ (design * info_w[:, None]) if isinstance(family, sm.families.Gaussian): scale = float(np.sum(info_w * (y_flat - mu) ** 2)) / (n - len(beta)) else: scale = 1.0 try: cov = scale * np.linalg.inv(xtwx) except np.linalg.LinAlgError as exc: raise ValueError( f"the glum fit's information matrix is singular ({exc}), so its " f"{len(beta)} coefficients are not identified. statsmodels " f"answers the same design with a pseudo-inverse, which picks one " f"arbitrary solution out of infinitely many.") from exc se = np.sqrt(np.clip(np.diag(cov), 0.0, None)) with np.errstate(divide='ignore', invalid='ignore'): z = np.where(se > 0, beta / se, 0.0) p_values = 2.0 * st.norm.sf(np.abs(z)) null_kwargs = {'family': family} if offset is not None: null_kwargs['offset'] = offset if weights is not None and isinstance(family, sm.families.Binomial): null_kwargs['var_weights'] = var_weights null_fit = sm.GLM(y_flat, np.ones((n, 1)), **null_kwargs).fit() names = [str(c) for c in columns] model = _AbsorbedDesign(y_flat, design, names) model.__class__ = _GLUM_DESIGN_CLASS llf = family.loglike(y_flat, mu, var_weights=var_weights, scale=scale) deviance = family.deviance(y_flat, mu, var_weights=var_weights) return _GlumResults( params=pd.Series(beta, index=names), bse=pd.Series(se, index=names), pvalues=pd.Series(p_values, index=names), resid=y_flat - mu, fitted=mu, scale=scale, df_model=len(beta) - 1, df_resid=n - len(beta), nobs=n, model=model, family=family, llf=llf, null_deviance=float(null_fit.null_deviance), llnull=float(null_fit.llf), deviance=deviance, n_iter=int(getattr(estimator, 'n_iter_', 0))) #: The design class name :mod:`spacr.regression_qc` resolves a GLM by. Its #: scale rule reads the dispersion off ``model.scale``, which is what #: :class:`_GlumResults` reports -- 1 for the fixed-dispersion families and #: the Pearson estimate for Gaussian, exactly as statsmodels does. _GLUM_DESIGN_CLASS = type('GLM', (_AbsorbedDesign,), {'__doc__': "The design of a glum-fitted GLM."})
[docs] def regression_model(X, y, regression_type='ols', groups=None, alpha=1.0, cov_type=None, weights=None, l1_ratio=0.5, quantile=0.5, hinge_threshold=None, huber_t=1.345, exposure=None, spline_knots=4, spline_degree=3, group_lasso_lambda='auto', rra_alpha=0.25, rra_permutations=10000, regression_backend=DEFAULT_REGRESSION_BACKEND, verbose=False, response_name="", transform="", glm_force_identity=False): """Dispatch to the requested regression backend and return the fitted model. Every name in :data:`REGRESSION_TYPES` is fittable here, and every one of them has a matching branch in :func:`process_model_coefficients`, so a model that fits can always be turned into a coefficient table. The backends, and what each is for: ================================== ======================================================== ``ols`` Ordinary least squares on a continuous well response. ``wls`` Weighted least squares; ``weights`` is the well's cell count, so a well of 400 cells outweighs one of 30. ``rlm``/``huber`` Robust M-estimation (Huber loss). For outlier-heavy wells: a handful of runaway wells no longer drag the fit. ``glm`` GLM with the family auto-selected from the response by :func:`pick_glm_family_and_link`. ``poisson`` Poisson GLM with a log link and ``offset(log(exposure))``, for per-well counts - so the coefficients are effects on the per-cell RATE, not on the well's headcount. ``quasi_binomial`` Binomial GLM whose dispersion is estimated from the Pearson chi-square, for overdispersed fractions. ``beta`` Beta regression, for a fraction strictly inside (0, 1). ``logit``/``probit`` GLM-binomial on a fraction, weighted by cell count. ``quantile`` Quantile regression at ``quantile``; fits the tail of the response rather than its mean. ``mixed`` Mixed-effects linear model with ``groups`` as the random intercept. ``lasso``/``ridge``/``elasticnet`` Penalised least squares. ``hinge`` Linear SVM (hinge loss) on a binarised response. ``horseshoe`` Sparse Poisson GLM with a horseshoe prior (spaCRPower's power-analysis model), via :mod:`spacr.power_model`. ``group_lasso`` Penalised least squares with a gene's guide columns penalised as ONE block, so a gene is selected or dropped as a set rather than one guide at a time - the penalised analogue of the mixed model's nesting, via :mod:`spacr.group_lasso`. ``rra`` MAGeCK-style robust rank aggregation: guides ranked by their marginal effect, aggregated to the gene BY RANK with a permutation P value, via :mod:`spacr.rra`. It forms no joint fit, so the collinearity and the p >> n width that constrain every backend above do not reach it. ================================== ======================================================== Settings a backend cannot read are REFUSED, not ignored — see :data:`REGRESSION_SETTINGS_USED`. :param X: Design matrix (DataFrame; column names become feature names). :param y: Response variable. :param regression_type: One of :data:`REGRESSION_TYPES`. :param regression_backend: WHO fits it -- one of :data:`REGRESSION_BACKEND_ORDER`. Default ``'statsmodels'``, which produced every existing result. A backend that cannot fit ``regression_type`` is REFUSED here with the reason, not ignored: the two controls constrain each other in both directions , and a settings CSV reaches this function without passing a panel that could have greyed the entry out. :param groups: Cluster identifiers for the mixed model. :param alpha: Penalty weight for ``lasso``/``ridge``/``elasticnet`` and the inverse SVM margin for ``hinge``; ``'auto'`` / ``None`` picks it by 5-fold cross-validation for all four (mean squared error for the penalised least-squares three, balanced accuracy for ``hinge``). :param cov_type: Covariance estimator for the likelihood fits (``'HC0'..'HC3'``); ``None`` for classical standard errors. :param weights: Per-observation weights - the well's cell count. Used as ``var_weights`` by ``logit``/``probit``/``quasi_binomial`` and as the WLS weights by ``wls``. :param l1_ratio: ``elasticnet`` mix; 1.0 is lasso, 0.0 is ridge. :param quantile: Quantile fitted by ``quantile`` regression, in (0, 1). :param hinge_threshold: Cut used to binarise a continuous response for ``hinge``; see :func:`binarise_response`. :param spline_knots: Knots per continuous covariate for ``spline``. :param spline_degree: Polynomial degree of that basis; 3 is cubic. :param huber_t: Huber tuning constant for ``rlm``/``huber``, in units of the estimated residual scale. 1.345 gives 95% efficiency under normality. :param exposure: Per-observation exposure (the well's cell count) used as ``offset(log(exposure))`` by ``horseshoe`` and by ``poisson`` (and by ``glm`` when it auto-selects a Poisson family). :param group_lasso_lambda: The block penalty for ``group_lasso``. Its own key rather than ``alpha`` because it is compared against :func:`spacr.group_lasso.max_lambda`, which is a property of the design, so a value carried over from a lasso run would mean something else here. :param rra_alpha: The top fraction of the guide ranking alpha-RRA aggregates over. MAGeCK's 0.25, which is what keeps a gene with one strong guide and three that did not cut findable. :param rra_permutations: Draws per distinct guide count in RRA's permutation null; 10,000 puts the smallest reportable P value at 1e-4. :returns: Fitted statsmodels / sklearn estimator. :raises ValueError: on an unsupported ``regression_type``, or when a setting the chosen backend cannot read was set to a non-default value. Example: .. code-block:: python import pandas as pd X = pd.DataFrame({'Intercept': 1.0, 'fraction': [0.1, 0.5, 0.9]}) model = regression_model(X, pd.Series([0.2, 0.4, 0.7]), 'ols') model.params['fraction'] # the recovered slope """ if regression_type in UNSUPPORTED_REGRESSION_TYPES: raise ValueError( f"Unsupported regression type {regression_type}: " f"{UNSUPPORTED_REGRESSION_TYPES[regression_type]}") if regression_type not in REGRESSION_TYPES: raise ValueError( f"Unsupported regression type {regression_type}. " f"Supported types: {list(REGRESSION_TYPES)}") y_flat = np.asarray(y, dtype=float).reshape(-1) use_auto_alpha = alpha is None or (isinstance(alpha, str) and alpha == 'auto') if _left_blank(cov_type): cov_type = None if _left_blank(hinge_threshold): hinge_threshold = None supplied = { 'alpha': 1.0 if use_auto_alpha else alpha, 'l1_ratio': l1_ratio, 'cov_type': cov_type, 'quantile': quantile, 'hinge_threshold': hinge_threshold, 'huber_t': huber_t, 'spline_knots': spline_knots, 'spline_degree': spline_degree, 'group_lasso_lambda': group_lasso_lambda, 'rra_alpha': rra_alpha, 'rra_permutations': rra_permutations, } backend = _require_backend(regression_type, regression_backend) _reject_unused_settings(regression_type, { name: (supplied[name], default) for name, default in _MODEL_LEVEL_DEFAULTS.items()}) def _find_best_alpha(model_cls): """Fit and return the requested cross-validated penalty estimator.""" alphas = np.logspace(-5, 5, 100) if model_cls == 'lasso': cv = LassoCV(alphas=alphas, cv=5, max_iter=10000).fit(X, y_flat) elif model_cls == 'ridge': cv = RidgeCV(alphas=alphas, cv=5).fit(X, y_flat) elif model_cls == 'elasticnet': cv = ElasticNetCV(alphas=alphas, l1_ratio=l1_ratio, cv=5, max_iter=10000).fit(X, y_flat) else: raise ValueError(f"_find_best_alpha called with unknown model_cls={model_cls!r}") print(f"Optimal alpha for {model_cls}: {cv.alpha_:.4g} " f"(MSE: {mean_squared_error(y_flat, cv.predict(X)):.4f})") return cv def _glm_binomial(link=None, scale=None): """Fit and return an optionally weighted binomial GLM for ``link`` and ``scale``.""" family = sm.families.Binomial(link=link) if link else sm.families.Binomial() kwargs = {'family': family} if weights is not None: kwargs['var_weights'] = np.asarray(weights).ravel() fit_kwargs = {} if scale is not None: fit_kwargs['scale'] = scale if cov_type is not None: fit_kwargs['cov_type'] = cov_type return sm.GLM(y, X, **kwargs).fit(**fit_kwargs) def _poisson_offset(): """``log(exposure)``, or None with a warning when there is no exposure. A per-well POSITIVE COUNT is not comparable between wells of different size: ``process_scores`` sums the response for the count models, so a well of 2000 cells contributes roughly four times the count of a well of 500 at the identical underlying rate. Modelling that count without ``offset(log(Ntotal))`` asks the covariates to explain well size, and any covariate correlated with it comes back as a hit. Measured on a 400-well simulation with a nuisance covariate that drives well size and nothing else, and a true rate coefficient of +1.5: without the offset the nuisance term came back at +1.88 with p = 0, ahead of the real effect; with it, +0.002 with p = 0.90. :returns: ``log(exposure)`` aligned with ``y``, or None. :raises ValueError: when the exposure is not positive and finite — ``log`` of it would be NaN/-inf and every downstream number would silently follow. """ if exposure is None: print("Warning: no per-well cell count reached the Poisson fit, so " "it models the raw count with no offset(log(cell_count)). " "Wells of different size are then not comparable and any " "covariate correlated with well size will look like a hit. " "Run the scores through process_scores so each well carries " "its cell count.") return None n_total = np.asarray(exposure, dtype=float).ravel() if n_total.size != np.asarray(y, dtype=float).reshape(-1).size: raise ValueError( f"the Poisson exposure has {n_total.size} entries but the " f"response has {np.asarray(y).reshape(-1).size}; each well " f"must carry its own cell count.") if not np.isfinite(n_total).all() or np.any(n_total <= 0): raise ValueError( "the Poisson exposure is the well's cell count, so it must be " f"finite and strictly positive; got " f"{np.nanmin(n_total)}-{np.nanmax(n_total)}. A well with no " f"cells has no rate to estimate and must be filtered out " f"(min_cells_per_well) rather than offset by log(0).") return np.log(n_total) def _glm_auto(): """Return a forced identity-link fit or the response-appropriate GLM.""" fit_y = y if glm_force_identity: family = sm.families.Gaussian(link=sm.families.links.Identity()) print(f" Using Gaussian family with Identity link for " f"{response_name or 'the response'}.") return sm.GLM(fit_y, X, family=family).fit( **({'cov_type': cov_type} if cov_type else {})) family = pick_glm_family_and_link(fit_y, name=response_name, transform=transform) if isinstance(family, sm.families.Poisson): _validate_poisson_response(fit_y, X) return sm.GLM(fit_y, X, family=family, offset=_poisson_offset()).fit( **({'cov_type': cov_type} if cov_type else {})) kwargs = {'family': family} if weights is not None and isinstance(family, sm.families.Binomial): kwargs['var_weights'] = np.asarray(weights).ravel() return sm.GLM(fit_y, X, **kwargs).fit( **({'cov_type': cov_type} if cov_type else {})) def _glm_poisson(): """Validate counts and return a Poisson-log fit with any exposure offset.""" _validate_poisson_response(y, X) family = sm.families.Poisson(link=sm.families.links.Log()) return sm.GLM(y, X, family=family, offset=_poisson_offset()).fit( **({'cov_type': cov_type} if cov_type else {})) def _wls(): """Validate captured per-well weights and return the fitted WLS model.""" if weights is None: raise ValueError( "regression_type='wls' needs per-well weights, and no " "'cell_count' column reached the model. Weighted least " "squares with unit weights is exactly OLS, so spaCR will not " "fit it under the 'wls' label. Use 'ols', or run the scores " "through process_scores so each well carries its cell count.") w = np.asarray(weights, dtype=float).ravel() if not np.isfinite(w).all() or np.any(w <= 0): raise ValueError( "WLS weights must be finite and positive (they are per-well " f"cell counts); got {np.nanmin(w)}-{np.nanmax(w)}.") return sm.WLS(y, X, weights=w).fit( **({'cov_type': cov_type} if cov_type else {})) def _rlm(): """Return a robust linear fit using the captured Huber tuning constant.""" return sm.RLM(y, X, M=sm.robust.norms.HuberT(t=huber_t)).fit() def _quantile(): """Validate the captured quantile and return its fitted regression.""" if not 0.0 < float(quantile) < 1.0: raise ValueError( f"quantile must lie strictly inside (0, 1); got {quantile!r}. " f"0.5 is the median fit.") return sm.QuantReg(y, X).fit(q=float(quantile)) def _hinge(): """Return a hinge classifier using auto-CV or fixed ``C = 1 / alpha``.""" y_binary = binarise_response(y, hinge_threshold, name='dependent variable') if use_auto_alpha: return _find_best_hinge_alpha(y_binary) strength = float(alpha) if strength <= 0: raise ValueError( f"alpha must be positive for hinge regression; got {alpha!r}.") model = _hinge_estimator(strength) model.fit(X, y_binary) return model def _hinge_estimator(strength): """A LinearSVC at regularisation ``strength`` (``C = 1 / strength``). ``class_weight='balanced'`` because a screen's positive class is routinely a small minority of wells: an unweighted hinge on a 95/5 split minimises its loss by calling every well negative, which returns a coefficient vector of ~0 for every gRNA and reads downstream as "no hits". Balancing reweights each class by its inverse frequency, so the decision boundary is fitted to separate the classes rather than to count them. """ return LinearSVC(C=1.0 / strength, loss='hinge', dual=True, max_iter=20000, random_state=0, class_weight='balanced') def _find_best_hinge_alpha(y_binary): """Pick the hinge penalty by stratified CV on balanced accuracy. The same 5-fold shape ``_find_best_alpha`` uses for the penalised least-squares backends. Balanced accuracy rather than accuracy: on an imbalanced screen plain accuracy is maximised by the degenerate all-negative fit, so scoring on it would cross-validate its way to the very failure ``class_weight='balanced'`` exists to prevent. Falls back to the unpenalised-scale default ``C = 1`` when the response has too few wells in a class to split five ways — a two-fold CV on three positive wells is noise, and choosing a penalty from noise is worse than not choosing one. """ from sklearn.model_selection import cross_val_score strengths = np.logspace(-3, 3, 13) minority = int(min(np.sum(y_binary == 0), np.sum(y_binary == 1))) n_splits = min(5, minority) if n_splits < 2: print(f"hinge: alpha='auto' needs at least two wells in each " f"class to cross-validate and the smaller class has " f"{minority}; falling back to alpha=1.") model = _hinge_estimator(1.0) model.fit(X, y_binary) return model folds = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=0) scores = [] for strength in strengths: with warnings.catch_warnings(): warnings.simplefilter('ignore') fold_scores = cross_val_score( _hinge_estimator(strength), X, y_binary, cv=folds, scoring='balanced_accuracy') scores.append(float(np.mean(fold_scores))) best = float(strengths[len(strengths) - 1 - int(np.argmax(scores[::-1]))]) print(f"Optimal alpha for hinge: {best:.4g} " f"(balanced accuracy {max(scores):.4f}, {n_splits}-fold)") model = _hinge_estimator(best) model.fit(X, y_binary) return model def _named_design(name): """The design's column names, or a refusal that says why they matter. Gene-aware backends recover the gene behind each predictor from the column name produced by patsy. Refuse an unnamed array because group lasso would otherwise treat every column as a separate gene and reduce to ordinary lasso under a different label. """ columns = getattr(X, 'columns', None) if columns is None: raise ValueError( f"regression_type={name!r} groups the design's columns by " f"gene and reads that grouping from the COLUMN NAMES, so it " f"needs a DataFrame design; a bare array has no names to " f"group by. Build the design with " f"dmatrices(..., return_type='dataframe'), which is what the " f"pipeline hands in.") return columns def _group_lasso(): """Fit the named gene-grouped design and return compatible results.""" from . import group_lasso as group_lasso_module columns = _named_design('group_lasso') design = np.asarray(X, dtype=float) blocks = _design_column_groups(columns) gene_terms = _level_term_mask(columns) if _left_blank(group_lasso_lambda) or ( isinstance(group_lasso_lambda, str) and group_lasso_lambda.strip().lower() == 'auto'): lam = group_lasso_module.choose_lambda( design, y_flat, blocks, required=gene_terms if gene_terms.any() else None) print(f"group_lasso_lambda='auto': cross-validated over " f"{group_lasso_module.PATH_POINTS} penalties down from " f"this design's ceiling of " f"{group_lasso_module.max_lambda(design, y_flat, blocks):.4g}" f", chose {lam:.4g}.") else: lam = float(group_lasso_lambda) beta, intercept, converged = group_lasso_module.fit( design, y_flat, blocks, lam=lam) if not gene_terms.any(): raise ValueError( "regression_type='group_lasso' penalises a GENE's guide " f"columns as one block, and none of this design's " f"{len(gene_terms)} columns is a gRNA or gene term " f"(columns: {[str(c) for c in columns][:6]}). Every column " f"would be its own block, which is ordinary lasso under " f"another name. It is fitted on the design prepare_formula " f"builds, whose terms are 'fraction:grna[...]' or " f"'gene_fraction:gene[...]'.") if not np.any(beta[gene_terms]): ceiling = group_lasso_module.max_lambda(design, y_flat, blocks) raise ValueError( f"group_lasso shrank every one of the " f"{int(gene_terms.sum())} gRNA/gene coefficients to exactly " f"zero at group_lasso_lambda={lam!r}, so the fit carries no " f"information about any gene. This design's " f"group_lasso.max_lambda -- the penalty above which nothing " f"at all survives -- is {ceiling:.4g}, and the gene blocks " f"empty well below it. Set group_lasso_lambda='auto' to " f"cross-validate it, or a small fraction of that ceiling to " f"choose it yourself. Or fit an unpenalised model ('ols') to " f"see the effect sizes the penalty is shrinking away.") if not converged: print(f"Warning: the group lasso did not reach its tolerance in " f"{group_lasso_module.MAX_ITERATIONS} sweeps. The " f"coefficients are the last iterate, not the solution; " f"treat the selection as provisional.") genes_in_design = {label for label, is_gene in zip(blocks, gene_terms) if is_gene} selected = {label for label, coefficient, is_gene in zip(blocks, beta, gene_terms) if is_gene and coefficient != 0} model = _GroupLassoResults(beta, intercept, blocks, lam, converged) mse = mean_squared_error(y_flat, model.predict(X)) print(f"Group lasso MSE: {mse:.4f}, lambda={lam:g}, " f"{len(selected)} of {len(genes_in_design)} gene blocks " f"selected ({int(np.sum(beta[gene_terms] != 0))} of " f"{int(gene_terms.sum())} gRNA columns).") return model def _rra(): """Rank marginal design slopes and return gene-level alpha-RRA results.""" from . import rra as rra_module columns = _named_design('rra') design = np.asarray(X, dtype=float) genes = [_gene_of_design_column(column) for column in columns] gene_terms = _level_term_mask(columns) if not gene_terms.any(): raise ValueError( "regression_type='rra' aggregates a GENE's guides by rank, " f"and none of this design's {len(genes)} columns is a " f"gRNA or gene term (columns: {[str(c) for c in columns][:6]}" f"). It is fitted on the design prepare_formula builds, " f"whose terms are 'fraction:grna[...]' or " f"'gene_fraction:gene[...]'.") centred = design - design.mean(axis=0) response = y_flat - float(y_flat.mean()) spread = (centred ** 2).sum(axis=0) moving = spread > 0 slopes = np.zeros(design.shape[1], dtype=float) slopes[moving] = (centred[:, moving].T @ response) / spread[moving] ranked = np.where(moving, slopes, np.nan) table = rra_module.rank_aggregate( ranked, genes, alpha=float(rra_alpha), direction='both', n_permutations=int(rra_permutations)) if not len(table) or 'p_neg' not in table.columns: raise ValueError( "regression_type='rra' ranked no guide: every gRNA column of " "this design is constant, so no guide has a marginal effect " "to rank. Check the fraction threshold - a design whose guide " "columns do not vary carries no information about any guide.") two_sided = np.minimum(1.0, 2.0 * np.minimum( table['p_neg'].to_numpy(dtype=float), table['p_pos'].to_numpy(dtype=float))) by_gene = dict(zip(table['gene'].astype(str), two_sided)) p_values = np.array( [by_gene.get(str(gene), np.nan) if gene is not None else np.nan for gene in genes], dtype=float) called = int(np.sum(two_sided <= 0.05)) print(f"RRA: {len(table)} genes aggregated from " f"{int(np.sum(moving & gene_terms))} ranked guides, " f"alpha={float(rra_alpha):g}, {int(rra_permutations)} " f"permutations per guide count; {called} genes at an " f"uncorrected two-sided p <= 0.05.") return _RRAResults(slopes, p_values, columns, table) def _horseshoe(): """Return a horseshoe-Poisson fit for the captured design and exposure.""" return _fit_horseshoe_poisson(X, y, exposure) def _spline(): """OLS on a design whose COVARIATES carry a spline basis. The guide columns are untouched, so one coefficient and one P value per guide survive. The volcano, hit list and attribution can therefore read the result with no special case. What becomes free to bend is the nuisance trend that the straight line was assuming away. A column is treated as a covariate when it is CONTINUOUS and is not a guide or gene term -- an indicator has nothing to bend through, and expanding one would spend degrees of freedom on nothing. """ from .nonparametric_fits import spline_design covariates = [] for name in getattr(X, "columns", []): label = str(name) if "grna[" in label or "gene[" in label or label == "Intercept": continue column = np.asarray(X[name], dtype=float) if np.unique(column).size > 4: covariates.append(name) design = (spline_design(X, covariates, knots=int(spline_knots), degree=int(spline_degree)) if covariates else X) fitted = (sm.OLS(y, design).fit(cov_type=cov_type) if cov_type else sm.OLS(y, design).fit()) return fitted model_map = { 'ols': lambda: sm.OLS(y, X).fit(cov_type=cov_type) if cov_type else sm.OLS(y, X).fit(), 'spline': _spline, 'wls': _wls, 'rlm': _rlm, 'huber': _rlm, 'glm': _glm_auto, 'poisson': _glm_poisson, 'quasi_binomial': lambda: _glm_binomial(link=sm.families.links.Logit(), scale='X2'), 'beta': lambda: BetaModel(endog=y, exog=X).fit(), 'logit': lambda: _glm_binomial(link=sm.families.links.Logit()), 'probit': lambda: _glm_binomial(link=sm.families.links.probit()), 'quantile': _quantile, 'mixed': lambda: perform_mixed_model( y, X, groups, regression_backend=regression_backend), 'lasso': lambda: _find_best_alpha('lasso') if use_auto_alpha else Lasso(alpha=alpha, max_iter=10000).fit(X, y_flat), 'ridge': lambda: _find_best_alpha('ridge') if use_auto_alpha else Ridge(alpha=alpha).fit(X, y_flat), 'elasticnet': lambda: _find_best_alpha('elasticnet') if use_auto_alpha else ElasticNet(alpha=alpha, l1_ratio=l1_ratio, max_iter=10000).fit(X, y_flat), 'hinge': _hinge, 'horseshoe': _horseshoe, 'group_lasso': _group_lasso, 'rra': _rra, } if backend == 'pyfixest': if cov_type is not None: raise ValueError( f"regression_backend='pyfixest' absorbs rowID and columnID, " f"and the HC1/HC2/HC3 corrections are computed from the full " f"model's leverage, which an absorbed fit never forms. It " f"reports classical standard errors only, so " f"cov_type={cov_type!r} would be a label on numbers that did " f"not come from it. Fit with " f"regression_backend='statsmodels' to use cov_type, or clear " f"cov_type to absorb.") if regression_type == 'wls' and weights is None: raise ValueError( "regression_type='wls' needs per-well weights, and no " "'cell_count' column reached the model. Weighted least " "squares with unit weights is exactly OLS, so spaCR will not " "fit it under the 'wls' label. Use 'ols', or run the scores " "through process_scores so each well carries its cell count.") try: model = _fit_absorbed_least_squares( X, y, weights=weights if regression_type == 'wls' else None, kind='WLS' if regression_type == 'wls' else 'OLS') except ValueError as nothing_to_absorb: if 'nothing to absorb' not in str(nothing_to_absorb): raise print(" regression_backend='pyfixest' has nothing to absorb " "on this design -- there are no rowID or columnID terms in " "it, so either model_plate_position=False removed them or " "the screen sits on one row and one column. Fitting with " "statsmodels instead: with no factors to project out the " "two backends compute the same numbers, so this is the " "same fit by the only route left, not a different model.") model = model_map[regression_type]() elif backend == 'glum': if cov_type is not None: raise ValueError( f"regression_backend='glum' reports the classical GLM " f"standard errors -- the inverse information matrix at a " f"fixed dispersion -- and has no HC0..HC3 estimator that " f"matches statsmodels', so cov_type={cov_type!r} would be a " f"label on numbers that did not come from it. Fit with " f"regression_backend='statsmodels' to use cov_type.") model = _fit_glum_glm(X, y, regression_type, weights=weights, exposure=exposure) else: model = model_map[regression_type]() if regression_type in ['glm', 'poisson']: print(fit_quality_note(model)) print(summary_for_console(model, verbose=verbose)) if regression_type in ['lasso', 'ridge', 'elasticnet']: mse = mean_squared_error(y_flat, model.predict(X)) coefs = np.asarray(model.coef_).ravel() n_nonzero = int(np.sum(coefs != 0)) print(f"{regression_type.capitalize()} regression MSE: {mse:.4f}, " f"non-zero coefficients: {n_nonzero} of {X.shape[1]}") if n_nonzero == 0: if use_auto_alpha: raise ValueError( f"{regression_type} with alpha='auto' cross-validated its " f"way to the empty model: every one of the " f"{X.shape[1]} coefficients is exactly zero, because no " f"gRNA predicted the held-out wells better than their mean " f"did. That is a null screen, not a misconfiguration - the " f"fit is refused rather than written out as '0 significant " f"gRNAs', which is what it would look like. Check the " f"dependent variable and the aggregation, or fit an " f"unpenalised model ('ols') to see the effect sizes the " f"penalty is shrinking away.") raise ValueError( f"{regression_type} shrank all {X.shape[1]} coefficients to " f"exactly zero at alpha={alpha!r}: the penalty is far larger " f"than the scale of this design, so the fit carries no " f"information about any gRNA. Lower alpha, or set it to " f"'auto' to choose it by cross-validation.") return model
def _fit_horseshoe_poisson(X, y, exposure): """Fit spaCRPower's sparse Poisson model through :mod:`spacr.power_model`. The model is the one ``spaCRPower/R/fit_model.R`` fits:: Npositive_w ~ Poisson(Ntotal_w * exp(b0 + sum_g b_g * log10expression_wg)) b_g ~ horseshoe(df = 10) i.e. a Poisson GLM with a log link, an ``offset(log(Ntotal))`` exposure and a horseshoe sparsity prior doing the variable selection. In spaCR's terms ``y`` is the per-well positive-object count (``process_scores`` sums the response for this type, as it does for ``'poisson'``), ``exposure`` is the well's cell count and ``X`` is the ordinary spaCR design. The import is deliberately lazy and inside the branch: the horseshoe fitter is a separate module, and neither the ordinary regressions nor anything else that imports :mod:`spacr.ml` should pay for it or fail without it. :param X: Design matrix. :param y: Per-well positive counts. :param exposure: Per-well total cell counts (the Poisson exposure). :returns: The fitted object returned by ``spacr.power_model.fit_horseshoe_poisson``, which must expose ``params`` and either ``pvalues`` or ``bse`` indexed like ``X.columns``. :raises ImportError: when :mod:`spacr.power_model` is not installed yet, naming the entry point this branch calls. :raises ValueError: when no exposure is available, or the returned object does not carry the coefficients this pipeline needs. """ if exposure is None: raise ValueError( "regression_type='horseshoe' fits Npositive ~ ... + " "offset(log(Ntotal)), so it needs the per-well cell count as the " "exposure, and no 'cell_count' column reached the model. Without " "it the counts of a 400-cell well and a 30-cell well would be " "compared as if the wells were the same size.") try: from .power_model import ModelData, fit_model, gather_model_estimate except ImportError as exc: raise ImportError( "regression_type='horseshoe' needs spacr.power_model, which is " "not present in this install. The branch calls " "spacr.power_model.prepare/ModelData + fit_model + " "gather_model_estimate; install or restore that module to use it." ) from exc counts = _validate_poisson_response( y, X, model="horseshoe (a sparse Poisson GLM over well counts)") n_total = np.asarray(exposure, dtype=float).ravel() if n_total.size != counts.size: raise ValueError( f"horseshoe exposure has {n_total.size} entries but the response " f"has {counts.size}; they are the same wells and must align.") if not np.isfinite(n_total).all() or np.any(n_total <= 0): raise ValueError( "horseshoe exposure (the well cell count) must be finite and " f"positive; got {np.nanmin(n_total)}-{np.nanmax(n_total)}. " "log(Ntotal) is undefined otherwise.") if np.any(counts > n_total): raise ValueError( "horseshoe needs Npositive <= Ntotal per well: the response is a " "count of positive objects and the exposure is how many objects " "were imaged, so a well cannot have more positives than cells. " f"{int(np.sum(counts > n_total))} well(s) break that.") design = np.asarray(X, dtype=float) columns = [str(c) for c in X.columns] constant = [name for name, column in zip(columns, design.T) if np.ptp(column) == 0] model_data = ModelData( wells=np.asarray(X.index), genes=np.asarray(columns, dtype=object), Npositive=counts, Ntotal=n_total, log10expression=design, unidentified_genes=tuple(constant), ) fit = fit_model(model_data, seed=0, standardize=True) return _HorseshoeResults(fit, gather_model_estimate(fit)) class _HorseshoeResults: """Adapt a :class:`spacr.power_model.PowerFit` to the results API spaCR reads. :func:`process_model_coefficients` wants ``params`` and ``pvalues`` indexed by design column; the horseshoe model reports posterior draws. The translation is stated rather than implied: * ``params`` is the posterior MEAN of each coefficient - a point estimate under shrinkage, not a maximum-likelihood one, so it is already pulled towards zero for terms the prior judges null. That is the whole purpose of the model and the reason its coefficients are not comparable in magnitude with the OLS ones. * ``pvalues`` is the two-sided posterior TAIL MASS, ``2 * min(P(beta > 0), P(beta < 0))``. It is not a frequentist p-value and no null hypothesis was tested to get it; it is reported under that name because every consumer downstream - the volcano plot, the hit table, ``-log10(p_value)`` - reads that column and would otherwise be given nothing. A term whose posterior sits entirely on one side of zero gets 0. * Unidentified terms (a constant column, or a gRNA present in every well at the same fraction) are DROPPED rather than reported as zero: the model could not estimate them, and a zero with a p-value would read as a tested null. :param fit: the ``PowerFit`` returned by ``power_model.fit_model``. :param estimates: the frame ``power_model.gather_model_estimate`` builds. """ def __init__(self, fit, estimates): """Validate the posterior table and expose identified coefficients.""" required = ('gene', 'mean', 'sd', 'prob_positive', 'identified') missing = [c for c in required if c not in estimates.columns] if missing: raise ValueError( f"spacr.power_model.gather_model_estimate returned columns " f"{list(estimates.columns)}; spaCR's coefficient table needs " f"{list(required)} and {missing} are absent.") self.fit = fit self.estimates = estimates identified = estimates[estimates['identified'].astype(bool)] index = pd.Index(identified['gene'].astype(str), name=None) self.params = pd.Series(identified['mean'].to_numpy(), index=index) self.bse = pd.Series(identified['sd'].to_numpy(), index=index) prob_positive = identified['prob_positive'].to_numpy(dtype=float) tail = 2.0 * np.minimum(prob_positive, 1.0 - prob_positive) self.pvalues = pd.Series(np.clip(tail, 0.0, 1.0), index=index) self.converged = bool(getattr(fit, 'converged', True)) if not self.converged: print("Warning: the horseshoe fit did not meet its own " "convergence criterion; treat the coefficients as " "provisional and re-run with more steps or a NUTS backend.") def summary(self): """Return the per-term posterior summary, for save_summary_to_file.""" return self.estimates class _GroupLassoResults: """Adapt :mod:`spacr.group_lasso` to the estimator API spaCR reads. ``coef_`` and ``predict`` are all :data:`_SKLEARN_COEF_TYPES`' branch of :func:`process_model_coefficients` and :func:`calculate_p_values` ask of a penalised fit, so the group lasso reports EXACTLY what ``lasso`` and ``elasticnet`` report -- one signed coefficient per design column, and a selection frequency attached by the run -- rather than a second convention of its own. WHY THE COEFFICIENT AND NOT ``gene_effects``' NORM. ``gene_effects`` answers with ``||b_g||_2``, one non-negative number per gene, which is the natural summary of a block but is not what the pipeline downstream of the fit is built on: the volcano's x axis, ``coefficient_threshold``'s control spread and the hit table's sign all read a SIGNED per-column effect. The block's own coefficients carry that sign, and because the block is zero or none of it is, ``||b_g||_2 > 0`` and "this gene has a non-zero coefficient" are the same statement -- so nothing is lost by tabling the coefficients, and a caller who wants the norm is one ``np.linalg.norm`` away from it: ``coef_`` and ``groups`` are both on this object, which is why ``groups`` is kept rather than discarded after the fit. :param coefficients: one coefficient per design column, in column order. :param intercept: the unpenalised intercept. :param groups: the group label of each column, from :func:`_design_column_groups`. :param lam: the penalty that was applied. :param converged: whether block coordinate descent met its tolerance. """ def __init__(self, coefficients, intercept, groups, lam, converged): """Store flattened coefficients and fitted group-lasso metadata.""" self.coef_ = np.asarray(coefficients, dtype=float).ravel() self.intercept_ = float(intercept) self.groups = list(groups) self.lam = float(lam) self.converged = bool(converged) def predict(self, X): """``X @ coef_ + intercept_``, the fit's prediction for a design. :func:`calculate_p_values` needs the residual, and the residual needs this. Taking ``np.asarray`` rather than relying on pandas' matmul keeps it working for a plain array as well as the DataFrame the pipeline hands in. """ return np.asarray(X, dtype=float) @ self.coef_ + self.intercept_ class _RRAResults: """Adapt :mod:`spacr.rra` to the results API :func:`process_model_coefficients` reads. ``params`` and ``pvalues``, the same two attributes every statsmodels fit and :class:`_HorseshoeResults` expose, so ``rra`` needs no branch of its own. What each one IS, stated rather than implied, because neither comes from a joint fit: * ``params`` is the guide's MARGINAL least-squares slope -- the slope of the response on that guide's column alone, one parameter estimated at a time. RRA's whole claim is that it never forms the joint fit (see :mod:`spacr.rra`), and this screen is 823 guides against 610 wells, where the joint fit is undefined. A marginal slope is defined for every column at any width and is the analogue of MAGeCK's per-guide log fold change, which is what alpha-RRA ranks. * ``pvalues`` is the guide's GENE's permutation P value, two-sided as ``min(1, 2 * min(p_neg, p_pos))`` -- the standard combination of the two one-sided permutation tests :func:`spacr.rra.rank_aggregate` reports. Taking whichever tail is smaller and NOT doubling would be a one-sided test chosen after seeing the data. A row that names no gene -- the intercept, the row/column dummies -- was never ranked, so its P value is NaN rather than 1.0: it was not tested, and a 1.0 would read as "tested and found null". :param scores: the marginal slope of each design column, in column order. :param p_values: the two-sided permutation P value of each column's gene. :param index: the design column names. :param genes: :func:`spacr.rra.rank_aggregate`'s per-gene table, kept whole so ``rho_neg``/``rho_pos`` and the direction split are not lost. """ def __init__(self, scores, p_values, index, genes): """Index marginal scores and permutation p-values by design column.""" feature_index = pd.Index([str(name) for name in index]) self.params = pd.Series(np.asarray(scores, dtype=float), index=feature_index) self.pvalues = pd.Series(np.asarray(p_values, dtype=float), index=feature_index) self.genes = genes def summary(self): """The per-gene RRA table, for :func:`save_summary_to_file`.""" return self.genes #: Regression types ``random_row_column_effects=True`` may be combined with. #: ``'ols'`` is here because it is the DEFAULT value of ``regression_type``, so #: it cannot be told apart from "the user never touched the model dropdown"; #: ``None`` means "choose from the response", which the mixed branch answers. #: Every other name is a deliberate choice that the mixed override would throw #: away. _RANDOM_EFFECTS_COMPATIBLE = (None, 'ols', 'mixed') def _reject_unused_run_settings(settings): """Refuse a post-fit setting the chosen model will never read. :func:`regression_model` polices the six knobs that reach the estimator, and raises before a wrong number can become a result. These three do not reach the estimator at all — they configure how :func:`perform_regression` turns coefficients into a hit list — so nothing was checking them, and ``lasso_selection_threshold=0.9`` on an OLS run passed through fifteen of the seventeen types in silence. ``regression_type=None`` is policed as strictly as a named one: :func:`check_distribution` only ever auto-selects ``logit``, ``beta``, ``quasi_binomial``, ``ols`` or ``glm``, none of which reads any of these, so "it might pick lasso" is not a reason to let them through. :param settings: The finished settings dict. :raises ValueError: naming the setting, the type and the alternative. """ reg_type = settings.get('regression_type', 'ols') _reject_unused_settings(reg_type, { name: (settings.get(name, default), default) for name, default in _RUN_LEVEL_DEFAULTS.items()}) return settings def _reconcile_random_row_column_effects(settings): """Make ``random_row_column_effects=True`` and ``regression_type`` agree. :func:`regression` reacts to the flag by fitting a MixedLM with row and column variance components, whatever ``regression_type`` says — and ``_perform_regression_set_paths`` had already named the results folder after ``settings['regression_type']``. A run configured as ``'lasso'`` with the flag on therefore fitted a mixed model and wrote it to ``results/<screen>/lasso/``, where nothing in the folder, the volcano filename or the settings CSV disagreed. Every penalty setting that run carried was ignored too, silently, because the mixed branch never reaches :func:`regression_model` and so never reaches :func:`_reject_unused_settings`. Two things happen here, both before any file is written: * an incompatible model choice is REFUSED, naming both settings; * a compatible one is rewritten to ``'mixed'`` in ``settings``, so the folder, the volcano filename and the saved settings all name the model that was actually fitted. :param settings: The finished settings dict; mutated in place. :raises ValueError: when the flag is combined with a named model that is not a mixed model, with ``model_plate_position=False``, or with a setting the mixed model cannot read. """ if not settings.get('random_row_column_effects', False): return settings if not settings.get('model_plate_position', True): raise ValueError( "random_row_column_effects=True fits rowID and columnID as " "variance components, and model_plate_position=False takes them " "out of the model entirely: there is nothing left for the mixed " "fit to make random. Set model_plate_position=True to fit plate " "position as variance components, or " "random_row_column_effects=False to leave it out.") reg_type = settings.get('regression_type', 'ols') if reg_type not in _RANDOM_EFFECTS_COMPATIBLE: raise ValueError( f"random_row_column_effects=True fits a mixed model with row and " f"column variance components, so it cannot also fit " f"regression_type={reg_type!r}: one of the two has to go. It used " f"to win silently, and the {reg_type!r} settings went with it — " f"the run wrote a MixedLM fit into results/<screen>/{reg_type}/ " f"and said nothing. Set random_row_column_effects=False to fit " f"{reg_type!r}, or regression_type='mixed' to fit the mixed model.") _reject_unused_settings('mixed', { 'alpha': (1.0 if settings.get('alpha') in (None, 'auto') else settings.get('alpha', 1.0), 1.0), 'l1_ratio': (settings.get('l1_ratio', 0.5), 0.5), 'cov_type': (settings.get('cov_type'), None), 'quantile': (settings.get('quantile', 0.5), 0.5), 'hinge_threshold': (settings.get('hinge_threshold'), None), 'huber_t': (settings.get('huber_t', 1.345), 1.345), 'spline_knots': (settings.get('spline_knots', 4), 4), 'spline_degree': (settings.get('spline_degree', 3), 3), }) if reg_type != 'mixed': print(f"random_row_column_effects=True: fitting 'mixed' rather than " f"{reg_type!r}, and naming the results folder for it.") settings['regression_type'] = 'mixed' return settings def _mixed_model_groups(df, dependent_variable, model_index, *, gene_column='gene'): """Return the outer random-intercept grouping for ``regression_type='mixed'``. The gene is the outer cluster and the guide is nested inside it. The model fits y ~ gene_fraction:gene + (1 | gene/grna) + rowID + columnID where ``(1 | gene/grna)`` is represented as ``groups=gene`` plus a guide variance component inside each gene. Row and column structure is carried by fixed terms, or by variance components when requested. See :func:`fit_mixed_model`. A single gene provides only one outer cluster and is refused. :param df: The cleaned long-format frame. :param dependent_variable: Response column name, named in the refusal so the message points at the run the user actually configured. :param model_index: Row index patsy kept, so the returned vector aligns with the design matrix row for row. :param gene_column: The outer grouping column. Default ``'gene'``. :returns: Series of gene ids, one per design row. :raises ValueError: when the screen has a single gene, naming the way out. """ genes = df.loc[model_index, gene_column] n_genes = genes.nunique() if n_genes > 1: print(f"Mixed model: grouping on {gene_column} ({n_genes} genes), " f"with guides nested inside. The gene sits above the guide, " f"which is the level the random effect describes.") return genes raise ValueError( f"a mixed model needs at least two clusters and this screen has one " f"{gene_column}. The random intercept has to sit above the guide, so " f"with a single gene there is nothing left for it to describe and " f"every guide BLUP against {dependent_variable!r} would be shrunk to " f"the same number. Fit a fixed-effects regression_type with " f"level='grna', which tests the guides directly and is the model a " f"one-gene screen supports.") def _write_regression_qc(model, X, y, df, dst, *, coef_df=None, regression_type=None, volcano_path=None): """Write the full QC suite for a fit into ``<dst>/regression_qc/``. Generate residual, scale-location, Q-Q, influence, collinearity, calibration, and coefficient diagnostics while the fitted design matrix and response are still available. Weights are deliberately not forwarded. ``regression`` passes cell counts to ``regression_model`` as ``var_weights`` / WLS weights / Poisson exposure for the types that take them, and for those types :func:`spacr.regression_qc.build_context` recovers the weights from the fitted model itself, so the hat diagonal, the residual and the scale agree. The unweighted types (ols, lasso, ridge, elasticnet) never saw the counts, and handing them in as ``weights`` would compute a weighted leverage for a fit nobody ran. The counts still reach the cell-count panel through ``metadata['cell_count']``, which is where that panel looks first. :param model: the fitted model. :param X: the design matrix that was fitted. :param y: the response that was fitted. :param df: the cleaned long-format frame, for the per-well metadata. :param dst: the run's results folder. :param coef_df: the coefficient table, so the p-value histogram shows the screen's p-values rather than the design's. :param regression_type: the spaCR regression type string. :param volcano_path: the volcano plot for this run, named on the report. :returns: the manifest dict, or ``None`` if the report could not be written. """ from .regression_qc import regression_qc_report metadata = None try: columns = [column for column in (schema.PLATE_KEY, schema.ROW_KEY, schema.COLUMN_KEY, schema.PRC_KEY, 'cell_count') if column in df.columns] if columns: metadata = df.loc[X.index, columns] if len(metadata) != len(X): raise ValueError( f"{len(metadata)} metadata rows for {len(X)} fitted rows; " f"the frame's index does not identify wells uniquely") except Exception as error: # noqa: BLE001 - advisory print(f"Regression QC: could not align per-well metadata to the fitted " f"rows ({type(error).__name__}: {error}); the plate/row/column " f"panels will skip rather than label the wrong well.") metadata = None try: return regression_qc_report( model, X, y, dst, metadata=metadata, coef_df=coef_df, regression_type=regression_type, volcano_path=volcano_path, verbose=True) except Exception as error: # noqa: BLE001 - advisory print(f"Regression QC report could not be written: " f"{type(error).__name__}: {error}") return None
[docs] def resolve_levels(regression_type, level='both'): """Which level(s) a run fits, given the backend and the ``level`` setting. ``mixed`` fits ONE model that already contains both levels -- the gene as a fixed effect and the guide as a random effect nested inside it -- so it ignores ``level`` entirely and answers ``('gene',)``. That is why the GUI greys the dropdown out rather than hiding it: the setting exists, but this model does not read it. Every other backend is fixed effects only and cannot nest, so it fits one level at a time and ``level`` chooses which. ``'both'`` is TWO FITS. :param regression_type: the backend name, or ``None`` (not yet chosen). :param level: ``'both'`` (default), ``'grna'`` or ``'gene'``. :returns: a tuple of levels to fit, in the order they are fitted. :raises ValueError: for a level that is not one of :data:`LEVEL_CHOICES`. """ key = str(level).strip().lower() if key not in LEVEL_CHOICES: raise ValueError( f"level={level!r} is not a model level. Choose one of " f"{LEVEL_CHOICES!r}: 'grna' fits y ~ fraction:grna + rowID + " f"columnID, 'gene' fits y ~ gene_fraction:gene + rowID + " f"columnID, and 'both' fits each of them SEPARATELY.") if regression_type == 'mixed': return ('gene',) if key == 'both': return ('grna', 'gene') return (key,)
def _wide_fixed_effect_design(df, dependent_variable, *, level, model_plate_position=True, block_screen=False, intercept='fitted'): """Build one fixed-effects row per independent well from long fractions. This is the explicit long-to-wide alternative to Patsy's historical long-row interaction formula. Each guide/gene becomes a numeric fraction column, absent predictors are zero, and the response and metadata must be constant within a well. The returned feature names keep spaCR's existing ``fraction:grna[...]`` / ``gene_fraction:gene[...]`` contract so every estimator and results consumer can use the same coefficient path. """ from .regression_layout import long_to_wide_regression_data from .schema import SCREEN_KEY if level == 'grna': predictor, value, prefix = 'grna', 'fraction', 'fraction:grna[' elif level == 'gene': predictor, value, prefix = 'gene', 'gene_fraction', 'gene_fraction:gene[' else: raise ValueError(f"wide model design needs one level, got {level!r}") metadata = [dependent_variable, 'plateID', 'rowID', 'columnID'] if 'cell_count' in df.columns: metadata.append('cell_count') if block_screen and SCREEN_KEY in df.columns: metadata.append(SCREEN_KEY) wide = long_to_wide_regression_data( df, index_columns='prc', predictor_column=predictor, value_column=value, metadata_columns=metadata, fill_value=0.0, ).sort_values('prc', kind='stable').reset_index(drop=True) identifiers = sorted( set(map(str, df[predictor].dropna().astype(str).unique())) ) missing = [name for name in identifiers if name not in wide.columns] if missing: raise AssertionError( "wide pivot lost predictor columns: " + ", ".join(missing[:5]) ) predictors = wide[identifiers].apply(pd.to_numeric, errors='raise').copy() predictors.columns = [prefix + name + ']' for name in identifiers] design_parts = [] mode = str(intercept or 'fitted').strip().lower() if mode not in INTERCEPT_MODES: raise ValueError( f"intercept={intercept!r} is not one of {list(INTERCEPT_MODES)}" ) if mode in ('fitted', 'control'): design_parts.append(pd.DataFrame({'Intercept': np.ones(len(wide))})) design_parts.append(predictors.reset_index(drop=True)) nuisance_columns = [] if model_plate_position: nuisance_columns.extend(['plateID', 'rowID', 'columnID']) if block_screen: nuisance_columns.append(SCREEN_KEY) for column in nuisance_columns: if column not in wide.columns: raise ValueError( f"model_data_layout='wide' needs nuisance column {column!r}" ) values = wide[column].astype(str) levels = sorted(values.unique()) for category in levels[1:]: design_parts.append(pd.DataFrame({ f'{column}[T.{category}]': values.eq(category).astype(float) })) X = pd.concat(design_parts, axis=1) y = pd.DataFrame({ dependent_variable: pd.to_numeric( wide[dependent_variable], errors='raise' ).to_numpy(dtype=float) }) return y, X, wide
[docs] def regression(df, csv_path, dependent_variable='predictions', regression_type=None, alpha=1.0, random_row_column_effects=False, nc='233460', pc='220950', controls=None, dst=None, cov_type=None, plot=False, l1_ratio=0.5, quantile=0.5, hinge_threshold=None, hinge_n_boot=200, huber_t=1.345, qc=True, spline_knots=4, spline_degree=3, legacy_volcano=False, level='grna', level_dst=None, draw_shared_panels=True, group_lasso_lambda='auto', rra_alpha=0.25, rra_permutations=10000, model_plate_position=True, regression_backend=DEFAULT_REGRESSION_BACKEND, verbose=False, transform="", intercept='fitted', intercept_value=0.0, model_data_layout='long'): """Run the full regression pipeline: clean, fit, extract coefficients, optional volcano plot. :param df: Long-format DataFrame with gRNA/gene fractions and the dependent variable. :param csv_path: Path used to derive the volcano-plot filename. :param dependent_variable: Response column name. Default ``'predictions'``. :param regression_type: Model type; auto-selected via :func:`check_distribution` when ``None``. :param regression_backend: WHO fits it, one of :data:`REGRESSION_BACKEND_ORDER`. Default ``'statsmodels'``. It is threaded to whichever fitter this run reaches -- the mixed branch and :func:`regression_model` alike -- so one setting answers for the whole run, and a backend that cannot fit the chosen family is refused by name before any design is built. :param alpha: Regularisation strength for penalised models. :param random_row_column_effects: If True, fit a mixed model with random row/column effects. :param model_plate_position: Whether ``plateID``, ``rowID`` and ``columnID`` are terms in a fixed-effects model (or the corresponding grouping/variance structure in a mixed model). Direct calls default to ``True`` for API compatibility; new application settings default to ``False`` so the terms are opt-in. See :func:`prepare_formula` for the measured costs of including or omitting them. ``False`` with ``random_row_column_effects=True`` is refused: there is nothing left to make random. :param nc: Negative-control gene identifier. Default ``'233460'``. :param pc: Positive-control gene identifier. Default ``'220950'``. :param controls: Explicit list of control identifiers. :param dst: Output directory for plots and summaries. :param cov_type: Optional covariance estimator for the likelihood fits. :param plot: If True, render the volcano plot after fitting. :param l1_ratio: ``elasticnet`` L1/L2 mix. :param quantile: Quantile fitted by ``quantile`` regression. :param hinge_threshold: Response cut used to binarise for ``hinge``. :param hinge_n_boot: Bootstrap resamples behind the hinge p-values. :param huber_t: Huber tuning constant for ``rlm``/``huber``. :param group_lasso_lambda: Block penalty for ``group_lasso``. :param rra_alpha: Top fraction of the guide ranking ``rra`` aggregates. :param rra_permutations: Draws per guide count in ``rra``'s null. :param qc: Write the regression QC suite into ``<dst>/regression_qc/``. :param legacy_volcano: also draw the ORIGINAL matplotlib volcano. Default ``False``. The interactive one is far faster and the house-style panel is what a run now produces; drawing both gives two volcanoes in two idioms on the same grid. Requires ``dst`` and a design matrix, so it is skipped for the mixed branch and when no destination was given. :param level: WHICH MODEL TO FIT -- ``'grna'`` (default) or ``'gene'``. One level, one design; ``'both'`` is refused here because it is two fits. :func:`regression_levels` is the entry point that does both. Ignored by ``regression_type='mixed'``, which fits the gene fixed and the guide random inside it and so is already both levels. :param level_dst: Where THIS LEVEL's figures go -- the QC suite, the volcano, the publication sheet. Defaults to ``dst``, which is what a single-level run wants. :func:`regression_levels` gives each level its own subfolder so two fits cannot overwrite each other's ``regression_figure.pdf``. :param draw_shared_panels: Draw the guide-fraction and response distributions, which describe the DATA and not the fit. False on the second of two fits, so the figure grid gets one copy rather than two identical ones. :param model_data_layout: ``'long'`` preserves the historical formula with one fitted row per well-guide pair. ``'wide'`` pivots fractions to one row per independent well before any fixed-effects estimator is fitted. Mixed models require their long nesting representation and therefore use long data even when a wide count input was supplied. :returns: ``(model, coef_df, regression_type)``. """ if controls is None: controls = [''] from .plot import volcano_plot, plot_histogram level_dst = dst if level_dst is None else level_dst volcano_path = create_volcano_filename( csv_path, regression_type, quantile if regression_type == 'quantile' else alpha, level_dst) if regression_type is None: regression_type = check_distribution(df[dependent_variable]) wanted = resolve_levels(regression_type, level) if len(wanted) != 1: raise ValueError( f"regression() fits ONE level; level={level!r} asks for " f"{list(wanted)}. Call regression_levels(), which fits each of " f"them separately and corrects each within itself.") level = wanted[0] model_layout = str(model_data_layout or 'long').strip().lower() if model_layout not in {'long', 'wide'}: raise ValueError( f"model_data_layout={model_data_layout!r}; choose 'long' or 'wide'." ) print(f"Using regression type: {regression_type}") dependent_variable, transform, glm_force_identity, conflict_note = ( resolve_glm_transform_conflict( dependent_variable, transform=transform, available=getattr(df, 'columns', ()), regression_type=regression_type)) if conflict_note: print(conflict_note) df = check_and_clean_data(df, dependent_variable) intercept_offset = 0.0 intercept_mode = str(intercept or 'fitted').strip().lower() if intercept_mode == 'control': df, intercept_offset = centre_on_controls(df, dependent_variable, nc) if intercept_offset: print(f"Intercept set to the negative controls: {dependent_variable} " f"centred by {intercept_offset:.6g}, so a coefficient reads " f"as its distance from {nc!r}.") else: print(f"Intercept left as fitted: no rows match " f"negative_control_id={nc!r}, so there is no control level to " f"centre on.") elif intercept_mode == 'value': intercept_offset = float(intercept_value or 0.0) if intercept_offset: df = df.copy() df[dependent_variable] = ( np.asarray(df[dependent_variable], dtype=float) - intercept_offset) print(f"Intercept pinned at {intercept_offset:.6g}: every " f"coefficient reads as its distance from that value.") qc_design = None fit_frame = None fit_counts = {} block_screen = screen_is_blockable(df) if block_screen: print(f"Blocking on {df['screenID'].nunique()} screens: " f"{sorted(df['screenID'].astype(str).unique())}") if regression_type == 'mixed' or random_row_column_effects: if model_layout == 'wide': print("model_data_layout='wide' was normalized back to long for " "the mixed model because guide-within-gene nesting is a " "long-data structure.") regression_type = 'mixed' level = 'gene' formula = prepare_formula( dependent_variable, random_row_column_effects=random_row_column_effects, block_screen=block_screen, level='gene', model_plate_position=model_plate_position, intercept=intercept) mixed_model, coef_df = fit_mixed_model( df, formula, level_dst, random_row_column_effects=random_row_column_effects, regression_backend=regression_backend) model = mixed_model observed_count = getattr(model, 'nobs', getattr(model, 'n_obs', None)) if observed_count is not None and np.isfinite(observed_count): fit_counts['n_rows_fitted'] = int(observed_count) inner = getattr(model, 'model', None) row_labels = getattr(getattr(inner, 'data', None), 'row_labels', None) if row_labels is not None and df.index.is_unique: fit_frame = df.loc[row_labels] elif fit_counts.get('n_rows_fitted') == len(df): fit_frame = df exog = getattr(inner, 'exog', None) if exog is not None: fit_counts['n_design_columns'] = int(exog.shape[1]) elif getattr(model, 'k_fe', None) is not None: fit_counts['n_design_columns'] = int(model.k_fe) fit_counts['layout'] = 'long' else: formula = prepare_formula(dependent_variable, random_row_column_effects=False, block_screen=block_screen, level=level, model_plate_position=model_plate_position, intercept=intercept) fit_df = df if model_layout == 'wide': y, X, fit_df = _wide_fixed_effect_design( df, dependent_variable, level=level, model_plate_position=model_plate_position, block_screen=block_screen, intercept=intercept, ) print(f"Model data pivoted to {len(fit_df)} independent well " f"rows and {X.shape[1]} design columns.") else: y, X = dmatrices(formula, data=df, return_type='dataframe') model_index = y.index if model_index.equals(fit_df.index): fit_frame = fit_df elif fit_df.index.is_unique: fit_frame = fit_df.loc[model_index] fit_counts = {'n_rows_fitted': int(len(y)), 'n_design_columns': int(X.shape[1]), 'layout': model_layout} if draw_shared_panels and not _show_well_distributions( df, dependent_variable, dst, plot=plot): plot_histogram(y, dependent_variable, dst=dst) plot_histogram(df, 'fraction', dst=dst) print('Data will not be scaled: the design is fractions and dummies ' 'on one common scale, and scaling it per column would rescale ' 'each gRNA coefficient by a different constant.') weights = (fit_df['cell_count'].loc[model_index] if 'cell_count' in fit_df.columns else None) groups = None print(f'Performing {regression_type} {level}-level regression') model = regression_model( X, y, regression_type=regression_type, groups=groups, alpha=alpha, cov_type=cov_type, weights=weights, l1_ratio=l1_ratio, quantile=quantile, hinge_threshold=hinge_threshold, huber_t=huber_t, spline_knots=spline_knots, spline_degree=spline_degree, exposure=weights, group_lasso_lambda=group_lasso_lambda, rra_alpha=rra_alpha, rra_permutations=rra_permutations, regression_backend=regression_backend, verbose=verbose, response_name=str(y.name) if hasattr(y, 'name') else '', transform=transform, glm_force_identity=glm_force_identity, ) fitted_exog = getattr(getattr(model, 'model', None), 'exog', None) if fitted_exog is not None: fit_counts['n_design_columns'] = int(fitted_exog.shape[1]) coef_df = process_model_coefficients( model, regression_type, X, y, nc, pc, controls, hinge_threshold=hinge_threshold, hinge_n_boot=hinge_n_boot) display(coef_df) qc_design = (X, y) if fit_frame is not None: contributing = fit_frame if fit_counts.get('layout') == 'wide' and 'prc' in fit_frame: contributing = df.loc[df['prc'].isin(fit_frame['prc'])] for name, column in (('n_wells', 'prc'), ('n_guides', 'grna'), ('n_genes', 'gene')): fit_counts[name] = int(contributing[column].nunique()) if plot and legacy_volcano: volcano_plot( coef_df, fold_change_col='coefficient', p_value_col='p_value', name_col='feature', x_transform='none', save_path=volcano_path, show=False, ) qc_manifest = None if qc and qc_design is not None and level_dst: qc_manifest = _write_regression_qc( model, qc_design[0], qc_design[1], fit_df, level_dst, coef_df=coef_df, regression_type=regression_type, volcano_path=volcano_path if plot else None) if level_dst: _show_house_style_panels(coef_df, plot=plot) _write_regression_sheet(coef_df, level_dst) try: from .figures.summary import summarise text = summarise(coef_df) if text: import textwrap print() print("SUMMARY") print(textwrap.fill(text, 88)) except Exception as error: # noqa: BLE001 - never lose a run over prose print(f"Could not summarise the run: {error}") coef_df = coef_df.copy() coef_df['level'] = level coef_df.attrs['fit_design'] = fit_counts if qc_manifest is not None and coef_df is not None: coef_df.attrs["qc_manifest"] = qc_manifest return model, coef_df, regression_type
[docs] def regression_levels(df, csv_path, dependent_variable='predictions', regression_type=None, level='both', dst=None, **kwargs): """Fit every level the run asked for, SEPARATELY, and return one per level. THIS IS THE TWO-FIT ENTRY POINT, and the reason it exists is that the one design spaCR used to fit cannot be fitted at all. ``gene_fraction`` is the SUM of the gene's gRNA fractions, so ``y ~ fraction:grna + gene_fraction:gene + plateID + rowID + columnID`` puts a block of columns and their own sums into one design. Measured on the reference TSG101 screen: 1248 parameters at rank 862 -- a 386-dimensional EXACT null space -- and the fit statsmodels returned had a residual sum of squares bit-identical to the one you get by adding seven times a null vector to it. See :data:`COLLINEAR_FORMULA_FRAGMENT`. Two fits, two tables, TWO CORRECTIONS. Each fit is its own multiple-testing family and is corrected within itself. Pooling them would be wrong twice over: they are not independent -- same wells, and the gene regressor IS the sum of the guide regressors -- and doubling the family size costs power for no protection. :func:`perform_regression` applies the correction per level and writes ``results_grna.csv`` and ``results_gene.csv``. ``regression_type='mixed'`` fits ONCE and returns one entry, ``'gene'``: that model has both levels inside it already, the gene as a fixed effect and the guide as a random effect nested in the gene. Its guide output is BLUPs, which is why it cannot be split into two testing families. :param df: long-format DataFrame of gRNA/gene fractions and the dependent variable, passed to :func:`regression` for every level. :param csv_path: path passed to :func:`regression`, which derives the volcano-plot filename from it. :param level: ``'both'`` (default), ``'grna'`` or ``'gene'``. :param dst: the run folder. With more than one fit each level's FIGURES go into ``<dst>/<level>/`` so they cannot overwrite each other; the tables stay in ``<dst>``, where every consumer looks for them. :param kwargs: passed straight through to :func:`regression`. :returns: ``dict`` mapping level to ``(model, coef_df, regression_type)``, in fit order. :raises ValueError: for a level that is not one of :data:`LEVEL_CHOICES`. """ import os levels = resolve_levels(regression_type, level) if regression_type == 'mixed' and str(level).strip().lower() != 'both': print(f"regression_type='mixed' fits the gene fixed with guides " f"random nested inside, so it is already both levels and " f"level={level!r} is not read. Its guide output is BLUPs, not " f"coefficients with p-values.") fits = {} for index, one in enumerate(levels): level_dst = dst if dst and len(levels) > 1: level_dst = os.path.join(str(dst), one) os.makedirs(level_dst, exist_ok=True) print(f"Fitting level {index + 1} of {len(levels)}: {one}") fits[one] = regression( df, csv_path, dependent_variable=dependent_variable, regression_type=regression_type, dst=dst, level=one, level_dst=level_dst, draw_shared_panels=(index == 0), **kwargs) regression_type = fits[one][2] return fits
def _show_well_distributions(frame, response_name, dst, plot=True): """Draw the guide-fraction and response distributions in the house style. :returns: True when they were drawn. False sends the caller back to the original ``plot_histogram``, because a figure is not worth losing a fit over. """ try: import matplotlib.pyplot as plt from .figures import distributions except Exception as error: # noqa: BLE001 print(f"Could not load the distribution panels: {error}") return False drawn = 0 per_panel = {"response": {"column": response_name}, "guide_fraction": {}} for key in distributions.ORDER: try: figure, panel = distributions.build_panel( key, frame, **per_panel.get(key, {})) except Exception as error: # noqa: BLE001 print(f"Distribution panel {key} did not draw: {error}") continue if not getattr(panel, "drawn", False): plt.close(figure) continue figure.set_label(panel.title) figure._spacr_title = panel.title if dst: try: from .plot import save_figure name = distributions.FILENAMES[key].format( response=response_name) save_figure(figure, os.path.join(str(dst), f"{name}.pdf"), bbox_inches="tight") except Exception: pass if plot: plt.show() plt.close(figure) drawn += 1 return drawn > 0 def _show_plates(frame, variable, dst): """Draw every plate as one small multiple. True when it was drawn.""" try: import matplotlib.pyplot as plt from .figures.plates import build_plates except Exception as error: # noqa: BLE001 print(f"Could not load the plate panel: {error}") return False try: figure, panel = build_plates(frame, variable, grouping="mean", min_max="allq", min_count=0) except Exception as error: # noqa: BLE001 print(f"The plate panel did not draw: {error}") return False if not getattr(panel, "drawn", False): plt.close(figure) return False figure.set_label(panel.title) figure._spacr_title = panel.title if dst: try: from .plot import save_figure save_figure( figure, os.path.join(str(dst), f"plate_heatmap_{variable}.pdf"), bbox_inches="tight") except Exception: pass plt.show() plt.close(figure) return True def _show_house_style_panels(coef_df, plot=True): """Draw each house-style panel and hand it to whatever is watching. One figure per panel rather than one sheet, because the grid puts each on its own lettered cell and a single composite would be one unreadable tile. The sheet is written to disk as well, for the version that goes in a paper. Never fatal: the fit is already done and losing a run over a figure would be the worst possible trade. """ if coef_df is None or not len(coef_df): return 0 try: import matplotlib.pyplot as plt from .figures import SHEET_ORDER, build_panel except Exception as error: # noqa: BLE001 print(f"Could not load the figure style: {error}") return 0 shown = 0 for key in SHEET_ORDER: try: figure, panel = build_panel(key, coef_df) except Exception as error: # noqa: BLE001 print(f"Panel {key} did not draw: {error}") continue if not panel.drawn: plt.close(figure) continue figure.set_label(panel.title) figure._spacr_title = panel.title if plot: plt.show() plt.close(figure) shown += 1 if shown: print(f"Drew {shown} regression panels in the house style.") return shown def _write_regression_sheet(coef_df, dst): """Write ``<dst>/regression_figure.pdf`` and its legend. Never fatal: a fit that produced a coefficient table has already done the work, and losing the run because a panel could not be drawn would be the worst possible trade. """ import os if coef_df is None or not len(coef_df): return None try: from .figures import build_sheet sheet = build_sheet(coef_df, width='double', target='print') folder = str(dst) os.makedirs(folder, exist_ok=True) path = os.path.join(folder, 'regression_figure.pdf') from .figure_sink import publish path = publish(sheet.figure, path, bbox_inches='tight') or path with open(os.path.join(folder, 'regression_figure_legend.txt'), 'w') as handle: handle.write(sheet.legend() + '\n') try: import matplotlib.pyplot as plt plt.close(sheet.figure) except Exception: pass print(f"Wrote the regression figure to {path} " f"({len(sheet.panels)} panels" + (f", {len(sheet.skipped)} not applicable" if sheet.skipped else '') + ').') return path except Exception as error: # noqa: BLE001 - never lose a run over a figure print(f"Could not draw the regression figure: {error}") return None #: What a run's statsmodels summary is written as, and every older name the #: reader still accepts. NEWEST FIRST -- the first one found wins. #: #: The name used to be ``mode_summary.csv``, which was wrong twice over: #: "mode" is a typo for "model", and the content is the statsmodels TEXT #: summary, never CSV. A name that does not follow the file is a path nobody #: can open, so the format is corrected going forward and the old names are #: still READ -- a run finished last month keeps its summary. #: How many coefficient rows the console will print before it stops and #: points at the file instead. The header of a statsmodels summary is about #: twenty lines; a screen's coefficient table is hundreds. CONSOLE_COEFFICIENT_LIMIT = 12
[docs] def fit_quality_note(model) -> str: """Return a one-line goodness-of-fit summary for a fitted GLM. McFadden's pseudo-R-squared compares log-likelihoods, and it is the appropriate summary for a GLM with a discrete response. A Gaussian identity-link fit instead reports ordinary R-squared because its likelihood is a density and the McFadden ratio is not interpretable on the usual zero-to-one scale. :param model: a fitted statsmodels GLM result. :returns: A labelled goodness-of-fit line for the console. """ family = getattr(model, 'family', None) if isinstance(family, sm.families.Gaussian): try: resid = np.asarray(model.resid_response, dtype=float).reshape(-1) observed = resid + np.asarray( model.fittedvalues, dtype=float).reshape(-1) centred = observed - observed.mean() total = float(np.dot(centred, centred)) residual = float(np.dot(resid, resid)) except (AttributeError, TypeError, ValueError): return "R²: not available for this fit" if not np.isfinite(total) or total <= 0: return "R²: not available for this fit (the response is constant)" return (f"R²: {1.0 - residual / total:.4f} (ordinary R², not " f"McFadden -- this is a Gaussian identity-link fit)") try: null_value = model.llnull if null_value is None: raise AttributeError("no null log-likelihood on this result") llf, null = float(model.llf), float(null_value) except (AttributeError, TypeError, ValueError): try: llf = float(model.llf) null = float(model.null_deviance) / -2.0 except (AttributeError, TypeError, ValueError): return "McFadden's R²: not available for this fit" if not np.isfinite(null) or null == 0: return "McFadden's R²: not available for this fit" return mcfadden_note(1.0 - (llf / null))
[docs] def mcfadden_note(r2) -> str: """Format McFadden's pseudo-R² and flag a negative value. A negative value means the fitted model predicts the response worse than an intercept-only model. The returned note explains that the coefficients should not be interpreted and points to a common cause: applying a response transform that duplicates the fitted family's link. :param r2: Pseudo-R² value, or a value convertible to ``float``. :returns: One-line diagnostic text suitable for a console or report. """ try: value = float(r2) except (TypeError, ValueError): return "McFadden's R²: not available for this fit" if value < 0: return ( f"McFadden's R²: {value:.4f} <-- NEGATIVE. This fit predicts the " f"response WORSE than its own intercept, so its coefficients and " f"P values do not describe the data. The usual cause is a " f"response transformed twice: check that `transform` is not " f"applying a log or logit that the family's link already applies." ) return f"McFadden's R²: {value:.4f}"
[docs] def summary_for_console(model, *, verbose=False, limit=CONSOLE_COEFFICIENT_LIMIT) -> str: """Return a statsmodels summary sized for terminal output. When the coefficient table exceeds ``limit``, the diagnostic header and notes are retained while the table is replaced by a pointer to the saved summary and the sortable Coefficients view. Set ``verbose=True`` to return the complete statsmodels rendering. :param model: Fitted model result with a ``summary()`` method. :param verbose: Return the complete summary regardless of its size. :param limit: Maximum coefficient rows printed in compact mode. :returns: Complete or compact plain-text model summary. """ try: text = str(model.summary()) except Exception as error: # noqa: BLE001 return (f"statsmodels could not render a summary for this fit " f"({type(error).__name__}: {error}).") if verbose: return text lines = text.splitlines() header = next((i for i, line in enumerate(lines) if "coef" in line and "std err" in line), None) if header is None: return text rule = next((i for i in range(header + 1, len(lines)) if set(lines[i].strip()) == {"-"}), None) if rule is None: return text end = next((i for i in range(rule + 1, len(lines)) if set(lines[i].strip()) == {"="}), len(lines)) rows = [line for line in lines[rule + 1:end] if line.strip()] if len(rows) <= limit: return text return "\n".join( lines[:rule + 1] + [f" {len(rows)} coefficients — not printed here. They are in the " f"run's model_summary.txt and in the Coefficients tab, which sorts " f"and filters them. Set verbose=True to print them."] + lines[end:])
SUMMARY_FILENAME = 'model_summary.txt' SUMMARY_FILENAMES = (SUMMARY_FILENAME, 'mode_summary.csv', 'summary.csv')
[docs] def save_summary_to_file(model, file_path=SUMMARY_FILENAME): """ Write ``model.summary().as_text()`` to ``file_path`` as plain text. The content is the statsmodels text summary, never CSV -- which is why the default name is :data:`SUMMARY_FILENAME` and no longer ``summary.csv``. Older runs on disk wrote ``mode_summary.csv``; every reader in this repository accepts both, see :data:`SUMMARY_FILENAMES`. :param model: Fitted statsmodels results object. :param file_path: Destination path. Default :data:`SUMMARY_FILENAME`. :returns: the path written, or ``None`` if there was nothing to write. NEVER RAISES INTO A FINISHED RUN. This is called after every table has been written; a backend whose ``summary()`` throws must not take the run down with it, and the caller is told by the ``None`` rather than by a traceback. """ summary = getattr(model, 'summary', None) if not callable(summary): return None try: summary_str = summary().as_text() except Exception as error: # noqa: BLE001 - a summary is not worth a run print(f"Could not render the model summary: " f"{type(error).__name__}: {error}") return None folder = os.path.dirname(os.path.abspath(file_path)) os.makedirs(folder, exist_ok=True) with open(file_path, 'w') as f: f.write(summary_str) return file_path
def _split_prc(text): """Return ``(plateID, rowID, columnID)`` for one ``prc`` well key. Parse from right to left because only the leading plate ID may contain the key separator. This preserves plate names such as ``'exp1_plate1'``. The row and column are returned exactly as they appear — nothing is canonicalised, because the caller rebuilds ``prc`` from these columns and a rewritten token would change the identity rows are joined on. Unescape the plate component to match :func:`spacr.schema.compose_prc` and :func:`spacr.schema.parse_prcf`; return row and column tokens unchanged. For keys with more than three components, require the final tokens to be a recognizable row/column pair. This accepts a plate containing the separator while rejecting a deeper ``prcf`` or ``prcfo`` key. Exactly three components remain accepted without positional-token validation. :param text: a ``prc`` key, e.g. ``'plate1_r1_c1'``. :returns: ``(plateID, rowID, columnID)``. :raises spacr.schema.KeyParseError: when ``text`` has fewer than three components, i.e. it is not a well key at all, or when it has more than three and the trailing pair is not a row and a column. """ key = str(text).strip() parts = key.split(schema.KEY_SEPARATOR) if len(parts) < 3: raise schema.KeyParseError( f'{text!r} is not a prc: expected plate_row_column, got ' f'{len(parts)} component(s).') plate = schema.KEY_SEPARATOR.join(parts[:-2]) row, column = parts[-2], parts[-1] if not plate.strip(): raise schema.KeyParseError( f'{text!r} is not a prc: it has no plate.') if not row.strip() or not column.strip(): raise schema.KeyParseError( f'{text!r} is not a prc: its row is {row!r} and its column is ' f'{column!r}, and an empty one identifies no well — every well of ' f'{plate!r} would be grouped together.') if len(parts) > 3 and not _is_row_column_pair(row, column): raise schema.KeyParseError( f'{text!r} is not a prc: it has {len(parts)} components and its ' f'last two, {row!r} and {column!r}, are not a row and a column. ' f'{_name_deeper_key(parts)}' f'If this really is a plate id containing ' f'{schema.KEY_SEPARATOR!r}, its row and column must be written ' f'the way spaCR writes them (r<N>/letters and c<N>/digits) for ' f'the plate to be separable from them.') return schema.unescape_filename_component(plate), row, column #: ``prc`` for a whole frame. :mod:`spacr.schema` owns it -- one place #: composes a key -- and this name is kept because seven call sites in this #: module use it. _compose_prc_column = schema.compose_prc_column def _is_row_column_pair(row, column): """True when ``(row, column)`` is recognisably a well's row and column. Deliberately narrow: it is the guard that stops :func:`_split_prc` from absorbing a ``prcf`` into an underscored plate id, so it must reject a ``(columnID, fieldID)`` pair and a ``(fieldID, objectID)`` pair. :param row: candidate ``rowID`` token. :param column: candidate ``columnID`` token. :returns: whether the pair can be a row and a column. """ row_text, column_text = str(row).strip(), str(column).strip() if not row_text or not column_text: return False if schema.is_positional_pair(row_text, column_text): return True if row_text[:1].lower() == schema.KEY_PREFIXES[schema.ROW_KEY]: row_ok = schema.row_index(row_text) is not None else: row_ok = schema.row_index_from_letters(row_text) is not None if not row_ok: return False if column_text[:1].lower() == schema.KEY_PREFIXES[schema.COLUMN_KEY]: return schema.column_index(column_text) is not None return column_text.isdigit() def _name_deeper_key(parts): """Return a sentence naming the deeper key ``parts`` looks like, or ''. Split out of :func:`_split_prc` only so the error it raises can say *which* mistake was made instead of describing the shape and leaving the caller to work it out. :param parts: the separator-split components of the rejected key. :returns: a sentence ending in a space, or ``''`` when the key does not look like a ``prcf`` / ``prcfo`` / timepoint key. """ tail = parts[-1] if schema.object_index(tail) is not None and len(parts) >= 5: return ('That is a prcfo (plate_row_column_field_object); ' '_split_prc takes a prc. Use schema.parse_prcfo. ') if schema.field_index(tail) is not None: return ('That is a prcf (plate_row_column_field); _split_prc takes a ' 'prc. Use schema.parse_prcf, or drop the field first. ') if schema.time_index(tail) is not None: return ('That ends in a timepoint; a prc has none. Aggregate the ' 'timepoints away before keying on the well. ') return '' def _qc_graph_type(fallback: str = 'jitter_bar') -> str: """The graph type the regression QC figures should start on. The DEFAULT GRAPH TYPE setting decides what is drawn FIRST, for every graph in spaCR and not only for Regression. The three QC figures below hardcoded ``'jitter_bar'`` and ignored it. THE FALLBACK IS THE OLD LITERAL, deliberately. :func:`graph_types. start_for` answers the CALLER'S OWN starting form when the user has expressed no preference, so a user who has set nothing sees exactly the figure they saw before. A preference nobody expressed must not move an existing view -- which is the rule ``fast_plots`` already follows for ``DEFAULT_MARK``. THE TWO VOCABULARIES ARE NOT THE SAME, and this is where they meet. ``graph_types`` stores ``'bar_jitter'``; :class:`spacr.plot.spacrGraph` draws ``'jitter_bar'``. :func:`graph_types.mark_for` is the translation, and skipping it hands matplotlib a name that its own error message lists as unknown. Every type that FITS ``categorical_continuous`` translates to something ``spacrGraph`` draws -- checked, all six. The shape is ``categorical_continuous`` because all three figures are one measurement grouped by ``plateID``. NO NOTE, BECAUSE NO COUNTS. :func:`graph_types.start_for` explains a swapped graph only when it is handed the per-group sizes, and these figures are drawn from a CSV path, not from sizes the caller holds. A note computed without counts is always empty, so the print that used to follow this call could never run. Passing counts is not the small fix it looks like: the fallback here is a MARK spelling, and on the too-thin path `start_for` re-checks it with `fits`, which a mark spelling fails -- see :func:`graph_types.mark_to_start_on`, which records exactly that. The swap concerns bar, box, violin and line drawn over 8 or fewer observations in a group, and these figures group whole plates of wells. :param fallback: what to draw when no preference is stored. The default is the literal these figures used before this existed. :returns: the graph type in ``spacrGraph``'s vocabulary. """ from .graph_types import mark_to_start_on return mark_to_start_on('categorical_continuous', fallback)[0] def _assign_prc_parts(df, column=schema.PRC_KEY, columns=schema.WELL_KEY_COLUMNS): """Split ``df[column]`` into plate / row / column and assign them onto ``df``. The frame-level counterpart of :func:`_split_prc`, and the ``prc`` sibling of :func:`_assign_prcfo_parts`. :param df: frame carrying ``column``. :param column: name of the ``prc`` column. Default ``'prc'``. :param columns: names to assign, in plate / row / column order. :returns: ``df``, mutated in place and returned for chaining. :raises spacr.schema.KeyParseError: when any value is not a ``prc``. """ parsed = [_split_prc(value) for value in df[column]] for position, name in enumerate(columns): df[name] = [part[position] for part in parsed] return df
[docs] def resolve_auto_inference(data, settings, *, well_column='prc', guide_column='grna'): """Choose ``analysis_mode`` for ``inference='auto'`` from the design. The simultaneous model estimates one coefficient per guide from the wells, so it needs more wells than guides -- with an intercept and any plate fixed effects on top -- before those coefficients are identifiable at all. Below that the design matrix is rank deficient: statsmodels still returns a number for every guide, but the numbers are one arbitrary solution out of infinitely many, and their P values describe nothing. That is not a hypothetical. The screen this was written for has 824 guides in 587 analysed wells; the published fit had 825 parameters, rank 579 and 8 residual degrees of freedom, and refitting it did not reproduce its own coefficients. ``auto`` therefore picks the permutation test whenever the simultaneous fit would be unidentifiable, and says so. It is deliberately conservative: it needs a real margin (``_IDENTIFIABILITY_MARGIN`` wells per guide) rather than a bare majority, because a design that only just fits is one dropped well away from not fitting. Anything other than ``inference='auto'`` is returned untouched, so an explicit choice is never overridden. :param data: the analysis table (a DataFrame); its distinct well and guide counts, and the permutation block column when present, size the design. :param settings: run settings; ``inference``, ``analysis_mode``, ``analysis_unit``, ``agg_type`` and ``guide_permutation_block`` are read. It is not modified. :param well_column: column whose distinct values count the wells. :param guide_column: column whose distinct values count the guides. :returns: ``(analysis_mode, reason)``. ``reason`` is a sentence naming the counts, suitable for the log and for the Methods section. """ inference = str(settings.get('inference', 'auto')).strip().lower() if inference != 'auto': return settings.get('analysis_mode', 'regression'), ( f"inference={inference!r} was set explicitly.") per_object = str(settings.get('analysis_unit') or 'well').lower() != 'well' aggregated = settings['agg_type'] if 'agg_type' in settings else 'mean' if per_object or aggregated is None: return 'regression', ( "auto chose the simultaneous model: the rows are one per OBJECT " "(agg_type is None or analysis_unit is not 'well'), and the " "permutation test needs one row per well. Set an agg_type such " "as 'mean' if the permutation test is wanted.") try: n_wells = int(data[well_column].nunique()) n_guides = int(data[guide_column].nunique()) except (KeyError, TypeError): return 'guide_permutation', ( "The design could not be measured, so the permutation test was " "used because it is valid regardless of the number of guides.") blocks = 0 block_column = str(settings.get('guide_permutation_block', 'plateID')) if block_column in getattr(data, 'columns', ()): blocks = max(int(data[block_column].nunique()) - 1, 0) parameters = 1 + blocks + n_guides required = parameters * _IDENTIFIABILITY_MARGIN if n_wells >= required: return 'regression', ( f"auto chose the simultaneous model: {n_wells} analysed wells for " f"{parameters} parameters ({n_guides} guides + intercept + " f"{blocks} block terms), at least the {_IDENTIFIABILITY_MARGIN}x " f"margin required.") return 'guide_permutation', ( f"auto chose the permutation test: {n_wells} analysed " f"wells cannot identify {parameters} simultaneous parameters " f"({n_guides} guides + intercept + {blocks} block terms). Each guide " f"is tested as a marginal association instead. Set " f"inference='parametric' to force the simultaneous fit.")
#: How many wells per estimated parameter ``auto`` insists on before it will #: choose the simultaneous model. 1.0 would accept a design with zero residual #: degrees of freedom, which fits perfectly and tests nothing. _IDENTIFIABILITY_MARGIN = 2.0 #: Fraction of count wells that must survive the score join before the run is #: allowed to continue. Below this the two inputs are describing different #: plates, and every number downstream is computed on whatever happened to #: overlap. _MINIMUM_PAIRED_WELL_FRACTION = 0.5
[docs] def normalize_regression_input_pairs(settings): """Return explicit ``score``/``count`` rows, migrating legacy lists. New settings store ``paired_data``. Older files remain valid: their flat lists are zipped positionally, exactly matching the former behaviour, and the migration is reported so the invisible legacy assumption is visible. :param settings: regression settings dictionary. ``paired_data`` is read when present, otherwise the legacy ``score_data`` and ``count_data`` lists; the dictionary is updated in place with the normalised ``paired_data`` and de-duplicated ``score_data``/``count_data`` lists. :returns: ``(pairs, migrated)``, where ``migrated`` is ``True`` when the rows came from the legacy lists. :raises ValueError: when ``paired_data`` is malformed or there is not at least one score path and one count path. """ from itertools import zip_longest rows = settings.get('paired_data') or [] migrated = False if rows: if not isinstance(rows, (list, tuple)): raise ValueError("paired_data must be a list of score/count rows") pairs = [] for index, raw in enumerate(rows): if not isinstance(raw, dict): raise ValueError(f"paired_data[{index}] must be a mapping") pairs.append({ 'score': raw.get('score') or raw.get('score_data'), 'count': raw.get('count') or raw.get('count_data'), 'plate': raw.get('plate') or raw.get('plateID'), 'database': raw.get('database') or raw.get('measurements'), }) for name in ('score_table', 'count_table'): if raw.get(name) is not None: pairs[-1][name] = raw[name] else: def paths(value): """Return ``value`` as a fresh path list, treating ``None`` as empty.""" if value is None: return [] return list(value) if isinstance(value, (list, tuple)) else [value] scores = paths(settings.get('score_data')) counts = paths(settings.get('count_data')) pairs = [ {'score': score, 'count': count, 'plate': None} for score, count in zip_longest(scores, counts) ] migrated = bool(pairs) if migrated: print("Legacy score_data/count_data lists were paired by position. " "Review and save the new paired_data table to make that " "relationship explicit.") if not pairs or not any(row['score'] for row in pairs) or not any( row['count'] for row in pairs): raise ValueError( "Regression needs at least one score CSV and one count CSV in " "paired_data.") settings['paired_data'] = pairs def unique(key): """Return truthy ``key`` paths once each in first-seen pair order.""" return list(dict.fromkeys( os.fspath(row[key]) for row in pairs if row.get(key))) settings['score_data'] = unique('score') settings['count_data'] = unique('count') return pairs, migrated
[docs] def load_regression_input_pairs(pairs): """Read paired inputs and resolve plate identity without filename guesses. Resolution order is own column, partner column, then pair-row order. Conflicting declarations are refused. Returns ``(count_frame, score_frame, audit_rows)``. :param pairs: sequence of mappings with ``'score'`` and ``'count'`` table paths (either may be empty), as returned by :func:`normalize_regression_input_pairs`. Each mapping's ``'plate'`` is overwritten with the resolved plate label. Database inputs also specify ``'score_table'`` or ``'count_table'``; tables from the same database are cached and paired separately. """ from .utils import correct_metadata score_frames = [] count_frames = [] seen_score_parts = set() seen_count_parts = set() audit = [] _parsed: dict = {} def read(path, table=None): """Return a cached corrected frame for ``path``, or ``None`` when blank.""" import time if not path: return None if table is not None: tabular._quote_identifier(table) path_key = (os.fspath(path) if tabular._backend_of(path) == 'postgres' else frame_handoff.key_for(path)) key = path_key if table is None else (path_key, table) if key not in _parsed: offered = frame_handoff.held(path) if table is None else None if offered is not None: note = frame_handoff.describe(path) print(f"Input {note}." if note else f"Input {os.path.basename(key)} handed over in memory.", flush=True) _parsed[key] = correct_metadata(offered) else: size = os.path.getsize(path_key) if os.path.exists(path_key) else 0 print(f"Reading {os.path.basename(path_key)} " f"({size / 1e6:.1f} MB)...", flush=True) started = time.time() options = {} if table is None else {'table': table} frame = correct_metadata(tabular.read_table( os.fspath(path), **options)) print(f" {len(frame):,} rows in " f"{time.time() - started:.1f} s.", flush=True) _parsed[key] = frame return _parsed[key] def plates(frame): """Return the frame's non-null plate identifiers as strings.""" if frame is None or 'plateID' not in frame.columns: return set() return {str(value) for value in frame['plateID'].dropna().unique()} for index, pair in enumerate(pairs): score = read(pair.get('score'), pair.get('score_table')) count = read(pair.get('count'), pair.get('count_table')) score_plates = plates(score) count_plates = plates(count) fallback = f'plate{index + 1}' if score_plates and count_plates: if score_plates == count_plates: resolved = score_plates rule = 'both files agree' elif count_plates < score_plates: score = score[score['plateID'].astype(str).isin(count_plates)] resolved = count_plates rule = 'matched score rows to count-file plate subset' elif score_plates < count_plates: count = count[count['plateID'].astype(str).isin(score_plates)] resolved = score_plates rule = 'matched count rows to score-file plate subset' else: raise ValueError( f"paired_data row {index + 1} conflicts: score file " f"declares {sorted(score_plates)}, count file declares " f"{sorted(count_plates)}. Pair files from the same " "plate.") elif score_plates: if (len(score_plates) > 1 and count is not None and fallback in score_plates): score = score[score['plateID'].astype(str) == fallback] resolved = {fallback} rule = ('assigned from pair row order; score file holds ' f'{len(score_plates)} plates') else: resolved = score_plates rule = 'copied from score file' elif count_plates: if (len(count_plates) > 1 and score is not None and fallback in count_plates): count = count[count['plateID'].astype(str) == fallback] resolved = {fallback} rule = ('assigned from pair row order; count file holds ' f'{len(count_plates)} plates') else: resolved = count_plates rule = 'copied from count file' else: resolved = {fallback} rule = 'assigned from pair row order' if (score is not None and count is not None and len(resolved) != 1 and (not score_plates or not count_plates)): raise ValueError( f"paired_data row {index + 1} cannot copy {sorted(resolved)} " "onto a partner with no plateID: one file contains several " "plates. Split that partner or give it an explicit plateID.") if score is not None and not score_plates: score = score.copy() score['plateID'] = next(iter(resolved)) if count is not None and not count_plates: count = count.copy() count['plateID'] = next(iter(resolved)) label = ', '.join(sorted(resolved)) pair['plate'] = label audit.append({'row': index + 1, 'plate': label, 'rule': rule, 'score': pair.get('score'), 'count': pair.get('count')}) print(f"Input pair {index + 1} ({label}): {rule}.") score_part = (os.fspath(pair.get('score')), pair.get('score_table'), tuple(sorted(resolved))) \ if score is not None else None count_part = (os.fspath(pair.get('count')), pair.get('count_table'), tuple(sorted(resolved))) \ if count is not None else None if score is not None and score_part not in seen_score_parts: score_frames.append(score) seen_score_parts.add(score_part) if count is not None and count_part not in seen_count_parts: count_frames.append(count) seen_count_parts.add(count_part) return (pd.concat(count_frames, ignore_index=True), pd.concat(score_frames, ignore_index=True), audit)
def _check_score_count_pairing(independent_df, dependent_df, merged_df, *, well_column='prc', record=None): """Fail loudly when the score and count tables describe different wells. The two inputs are never paired file-to-file: each list is concatenated and the two are joined on ``prc`` (``plateID_rowID_columnID``). So the plate ID is the pairing key, and a plate ID that differs by one character between the two sides silently produces an empty join. That is not hypothetical. A legacy score CSV carries its plate in a ``plate`` column stamped ``pplate1``, while the sequencing counts carry ``plate1``. Before :func:`spacr.utils.correct_metadata` was fixed to normalise that after the legacy promotion, the join returned zero rows and the run continued for another two hundred lines before dying inside a plot with ``KeyError: 0`` -- an error naming neither the plates, the files, nor the join. :param independent_df: Count-table rows before the score/count join. :param dependent_df: Score-table rows before the score/count join. :param merged_df: Rows retained by the score/count join. :param well_column: Column containing the unique well identifier. :param record: Optional mutable mapping that receives the matched and unmatched well counts for the persisted run summary. :raises ValueError: when the join is empty, or retains less than :data:`_MINIMUM_PAIRED_WELL_FRACTION` of the smaller input's wells. """ def _plates(frame): """Return sorted plate prefixes parsed from the frame's well column.""" if well_column not in frame.columns: return [] return sorted(frame[well_column].astype(str).str.split('_').str[0] .dropna().unique()) count_wells = independent_df[well_column].nunique() if \ well_column in independent_df.columns else 0 score_wells = dependent_df[well_column].nunique() if \ well_column in dependent_df.columns else 0 score_plates = _plates(dependent_df) count_plates = _plates(independent_df) matched = merged_df[well_column].nunique() if \ well_column in merged_df.columns else 0 comparable = min(score_wells, count_wells) if comparable and matched / comparable >= _MINIMUM_PAIRED_WELL_FRACTION: unused_counts = count_wells - matched unused_scores = score_wells - matched if record is not None: record["wells_paired"] = int(matched) record["wells_unpaired_counts"] = int(unused_counts) record["wells_unpaired_scores"] = int(unused_scores) if unused_counts or unused_scores: paired_label = "well" if matched == 1 else "wells" count_label = "well" if unused_counts == 1 else "wells" score_label = "well" if unused_scores == 1 else "wells" print( f"Paired {matched} {paired_label}. {unused_counts} " f"count-table {count_label} and {unused_scores} score-table " f"{score_label} had no matching identifier and were " f"excluded from the " f"regression.") return shared = sorted(set(score_plates) & set(count_plates)) detail = ( f"score wells: {score_wells} on plates {score_plates}\n" f" count wells: {count_wells} on plates {count_plates}\n" f" shared plates: {shared or 'NONE'}\n" f" paired wells: {matched}" ) if matched == 0: raise ValueError( f"The score and count tables have no well in common, so the " f"regression has nothing to fit.\n\n" f" {detail}\n\n" f"They are joined on prc = plateID_rowID_columnID, so the plate " f"ID is what pairs them -- the ORDER you listed the files in does " f"not matter, and the two lists need not be the same length. Make " f"the plate IDs agree: give every input a plateID column with " f"matching values, or state the pairing explicitly.") raise ValueError( f"Only {matched} of {comparable} pairable wells " f"({matched / comparable:.1%}) found a partner, which is below the " f"{_MINIMUM_PAIRED_WELL_FRACTION:.0%} required. The two inputs are " f"probably describing different plates or different well layouts.\n\n" f" {detail}\n\n" f"Continuing would fit the model on whichever wells happened to " f"overlap and report it as the whole screen.") def _identifiability_warning(data, settings, *, well_column='prc', guide_column='grna', level='grna'): """Warn when a fit is about to be run on too few wells. Returns the warning text, or ``None`` when the design is fine. Kept separate from :func:`resolve_auto_inference` because this one never changes what runs -- it only makes sure the user cannot miss what they are about to get. Count terms at the requested fit level because guide- and gene-level designs can have different widths. Return a warning only when the number of estimated intercept, block, and identifier terms is at least the number of analyzed wells. :param level: ``'grna'`` (default) or ``'gene'`` -- which fit is about to run, and therefore which identifiers are the parameters. """ identifier = 'gene' if str(level).strip().lower() == 'gene' else guide_column try: n_wells = int(data[well_column].nunique()) n_terms = int(data[identifier].nunique()) except (KeyError, TypeError): return None blocks = 0 block_column = str(settings.get('guide_permutation_block', 'plateID')) if block_column in getattr(data, 'columns', ()): blocks = max(int(data[block_column].nunique()) - 1, 0) parameters = 1 + blocks + n_terms if n_wells > parameters: return None return ( "\n" " ###############################################################\n" " # WARNING: this fit is saturated or not identifiable. #\n" " ###############################################################\n" f" {n_wells} analysed wells are being used to estimate " f"{parameters} parameters\n" f" ({n_terms} {identifier}s + intercept + {blocks} block terms).\n" "\n" " With at least as many parameters as wells, the model has no\n" " residual degrees of freedom and may also be rank deficient.\n" " Individual guide coefficients, standard errors and P values\n" " cannot be interpreted reliably.\n" "\n" " Set inference='nonparametric' to test each guide as a\n" " marginal association, wells reshuffled within each plate,\n" " coefficients simultaneously, or inference='auto' to let spaCR\n" " choose. The design\n" " diagnostics written beside the results show the rank, the\n" " residual degrees of freedom and the collinear guide pairs.\n") def _usable_nuisance_columns(data, settings) -> list: """The nuisance columns that are actually in the frame, said out loud. `guide_nuisance_columns` defaults to row and column, which every spaCR screen has and an imported table might not. `_nuisance_design` raises on an absent column -- correct for one the user typed, wrong for one that arrived as a default -- so the filtering happens here. SAID, NOT SILENT. A user who believes position was removed and reads a p-value computed without removing it has been told something false by omission, and the exchangeability the permutation rests on is exactly what those columns were there to protect. """ wanted = [str(c) for c in (settings.get('guide_nuisance_columns') or [])] if not wanted: return [] have = set(map(str, getattr(data, 'columns', ()))) usable = [c for c in wanted if c in have] missing = [c for c in wanted if c not in have] if usable: from .guide_permutation import _nuisance_design block = str(settings.get('guide_permutation_block', 'plateID')) while usable: try: _nuisance_design(data, block, usable) break except ValueError as exc: if "rank deficient" not in str(exc): break dropped = usable.pop() print(f"■ guide_nuisance_columns: {dropped!r} is collinear " f"with {block!r} on this layout -- every level of one " f"determines a level of the other -- so it cannot be " f"removed separately. Dropped; {block!r} already " f"absorbs it.") except Exception: # noqa: BLE001 break if missing: print(f"■ guide_nuisance_columns named {len(missing)} column(s) this " f"table does not have: {', '.join(missing)}. They are not " f"removed before the permutation, so any structure they carry " f"stays in the residual the shuffle treats as noise.") return usable def _report_exchangeability(data, outcome_column, settings, destination): """Measure and report whether the within-block shuffle is defensible. A COURTESY, NOT A PRECONDITION -- the same rule the montage pre-flight follows. It must never be the reason a run that produced results fails to report them, so every step is inside the guard. :param data: the merged per-well table. :param outcome_column: one phenotype column name. :param settings: the run's settings. :param destination: the run's results folder. When given, the report and a residual-by-position figure per block are written to its ``regression_qc`` folder, as a parametric run's QC is; the report's ``'qc'`` key holds what was written. :returns: the :func:`spacr.permutation_qc.block_residual_report`, or ``None`` when the check could not run. """ try: from .guide_permutation import (_nuisance_design, _residualize, prepare_long_guide_data) from .permutation_qc import (block_residual_report, exchangeability_verdict, write_permutation_qc) block = str(settings.get('guide_permutation_block', 'plateID')) nuisance = _usable_nuisance_columns(data, settings) wanted = list(dict.fromkeys([*nuisance, 'rowID', 'columnID'])) present = [c for c in wanted if c in getattr(data, 'columns', ())] _f, outcomes, _m = prepare_long_guide_data( data, outcome_column, block_column=block, nuisance_columns=present) y = pd.to_numeric(outcomes[outcome_column], errors='coerce').to_numpy(dtype=float) basis, _r = np.linalg.qr( _nuisance_design(outcomes, block, nuisance), mode='reduced') residuals = _residualize(y, basis) positions = {c: outcomes[c] for c in present if c in outcomes.columns and c != block} report = block_residual_report( residuals, outcomes[block], positions) verdict = exchangeability_verdict(report) if destination: try: report['qc'] = write_permutation_qc( destination, outcome_column, residuals, outcomes[block], positions, report, verdict, removed=[c for c in nuisance if c != block]) if report['qc'].get('figure'): print(f"Permutation QC written to {report['qc']['dir']}") except Exception as error: # noqa: BLE001 print(f"Permutation QC could not be written: " f"{type(error).__name__}: {error}") if verdict['ok']: print(f"Exchangeability: nothing found. Durbin-Watson " f"{report['durbin_watson']:.2f} over {report['n']:,} well(s) " f"in {report['blocks']} block(s), and no position column " f"explains the residual.") return report print("■ Exchangeability: the within-block shuffle is questionable.") for finding in verdict['findings'][:4]: print(f" {finding}") if verdict['remedy']: print(f" -> {verdict['remedy']}") return report except Exception: # noqa: BLE001 LOG.debug("could not report exchangeability", exc_info=True) return None
[docs] def resolve_regression_src(requested, automatic): """Resolve the root directory used for regression output. A blank ``requested`` value selects ``automatic``. An existing requested directory is used directly. If only the final path component is missing, that directory is created; missing parent directories are never created. A requested file, an unavailable parent, or a directory-creation error returns the automatic location with an explanatory message. :param requested: Requested output directory, or ``None``/blank to use the automatic location. :param automatic: Existing fallback directory, normally the directory containing the first count table. :returns: A ``(path, message)`` tuple. ``message`` is ``'automatic'`` when no override was requested; otherwise it describes the selected directory or the reason for falling back. """ if not isinstance(requested, str) or not requested.strip(): return automatic, 'automatic' wanted = os.path.abspath(os.path.expanduser(requested.strip())) if os.path.isdir(wanted): return wanted, f"Regression output directory: {wanted}." if os.path.exists(wanted): return automatic, ( f"The configured regression output path {wanted} is not a " f"directory. Results will be written to the automatic location " f"{automatic}.") parent = os.path.dirname(wanted) if os.path.isdir(parent): try: os.mkdir(wanted) except OSError as error: return automatic, ( f"The regression output directory {wanted} could not be " f"created ({error.strerror or type(error).__name__}). " f"Results will be written to the automatic location " f"{automatic}.") return wanted, f"Created regression output directory: {wanted}." return automatic, ( f"The regression output directory {wanted} was not created because " f"its parent directory {parent} does not exist. Results will be " f"written to the automatic location {automatic}.")
def _run_guide_permutation_analysis(data, outcome, destination, settings): """Run and persist the marginal guide analysis. This is the ``perform_regression`` branch used when ``analysis_mode='guide_permutation'``. Keeping it as a top-level function makes the correction and output contract testable without replaying score aggregation and sequencing QC. :returns: The long results, selected support family, significant rows, and a mapping of every artifact written by the analysis. """ from .guide_permutation import ( analyse_long_guide_table, plot_guide_permutation_volcano, save_guide_permutation_results, ) thresholds = settings.get('guide_min_wells', [1, 2, 3, 4]) if isinstance(thresholds, (int, np.integer)): thresholds = [int(thresholds)] thresholds = sorted({int(value) for value in thresholds}) if not thresholds or any(value < 1 for value in thresholds): raise ValueError('guide_min_wells must contain positive integers') primary = settings.get('guide_primary_min_wells') primary = thresholds[0] if _left_blank(primary) else int(primary) if primary not in thresholds: raise ValueError( f'guide_primary_min_wells={primary} is not in ' f'guide_min_wells={thresholds}') destination = os.path.abspath(os.path.expanduser(os.fspath(destination))) os.makedirs(destination, exist_ok=True) outcomes = [outcome] if isinstance(outcome, str) else list(outcome) missing = [column for column in outcomes if column not in data.columns] if missing: raise ValueError( f"dependent_variable names {missing} which are not columns of the " f"merged table. Available: {sorted(data.columns)[:20]}") per_object = str(settings.get('analysis_unit', 'well')).lower() != 'well' unaggregated = settings.get('agg_type') is None if per_object or unaggregated: why = (f"analysis_unit={settings.get('analysis_unit')!r}" if per_object else f"agg_type is None (regression_type=" f"{settings.get('regression_type')!r} fits objects)") raise ValueError( f"analysis_mode='guide_permutation' tests each guide across " f"WELLS, so it needs one row per well -- but " f"{why} gives one row " f"per object, and a well's phenotype then has many values. Set " f"analysis_unit='well' (with an agg_type such as 'mean'), or " f"choose analysis_mode='regression', which can model objects.") results = analyse_long_guide_table( data, outcomes, min_wells=thresholds, block_column=str(settings.get('guide_permutation_block', 'plateID')), nuisance_columns=_usable_nuisance_columns(data, settings), n_permutations=int(settings.get('guide_permutations', 200000)), random_state=int(settings.get('guide_permutation_seed', 0)), multiple_testing=str(settings.get('multiple_testing_method', 'fdr_bh')), alpha=float(settings.get('fdr_alpha', 0.05)), presence_threshold=float(settings.get('guide_presence_threshold', 0.0)), batch_size=int(settings.get('guide_permutation_batch_size', 500)), statistic=str(settings.get('grna_statistic', 'pearson')), ) for outcome_column in outcomes: _report_exchangeability(data, outcome_column, settings, destination) results = results.copy() results['grna'] = results['guide'] results['feature'] = ( 'fraction:grna[' + results['guide'].astype(str) + ']') results['coefficient'] = results['standardized_marginal_effect'] results['p_value'] = results['permutation_p_value'] results['q_value'] = results['adjusted_p_value'] results['condition'] = label_control_condition( results['feature'], results['grna'], nc=settings.get('negative_control_id'), pc=settings.get('positive_control_id'), controls=settings.get('nontargeting_control_grnas')) from .thresholds import coefficient_threshold control_effects = results.loc[ (results['minimum_wells_threshold'] == primary) & results['condition'].isin(('nc', 'control')), 'coefficient'] effect_threshold, effect_rule = coefficient_threshold( control_effects, method=settings.get('threshold_method', 'std'), multiplier=settings.get('threshold_multiplier', 3.0), centre=None) print(f"Effect-size cut: {effect_rule}") results['effect_size_threshold'] = ( np.nan if effect_threshold is None else float(effect_threshold)) results['passes_effect_size'] = ( True if effect_threshold is None else results['coefficient'].abs() >= float(effect_threshold)) paths = dict(save_guide_permutation_results( results, destination, prefix='guide_permutation')) if settings.get('guide_permutation_plot', True): single = len(outcomes) == 1 for response in outcomes: for threshold in thresholds: have = results.loc[ (results['outcome'] == response) & (results['minimum_wells_threshold'] == int(threshold))] if have.empty: print(f"No guide reached {threshold} well(s) for " f"{response!r}, so that panel of the " f"guide_min_wells sweep is not drawn. The " f"thresholds that did have guides are unaffected.") continue for suffix in ('pdf', 'png'): stem = (f'guide_permutation_min_{threshold}_wells' if single else f'guide_permutation_{response}_min_' f'{threshold}_wells') key = (f'plot_min_{threshold}_{suffix}' if single else f'plot_{response}_min_{threshold}_{suffix}') paths[key] = plot_guide_permutation_volcano( results, outcome=response, minimum_wells=threshold, save_path=os.path.join( destination, f'{stem}.{suffix}'), effect_threshold=effect_threshold, effect_threshold_label=effect_rule, ) try: from .guide_permutation import prepare_long_guide_data from .regression_diagnostics import write_diagnostic_suite fractions, well_outcomes, _metadata = prepare_long_guide_data( data, outcomes, block_column=str(settings.get('guide_permutation_block', 'plateID')), nuisance_columns=list(settings.get('guide_nuisance_columns') or [])) for response in outcomes: family = results.loc[ (results['outcome'] == response) & (results['minimum_wells_threshold'] == primary)] written = write_diagnostic_suite( os.path.join(destination, 'diagnostics'), fractions=fractions, block=well_outcomes[ str(settings.get('guide_permutation_block', 'plateID'))], p_values=family['permutation_p_value'].to_numpy(), adjusted=family['adjusted_p_value'].to_numpy(), alpha=float(settings.get('fdr_alpha', 0.05)), label=response if len(outcomes) > 1 else '', presence_threshold=float( settings.get('guide_presence_threshold', 0.0)), ) prefix = f'{response}_' if len(outcomes) > 1 else '' paths.update({f'{prefix}{key}': value for key, value in written.items()}) except Exception as error: # noqa: BLE001 - diagnostics are advisory print(f"Regression diagnostics were skipped: " f"{type(error).__name__}: {error}") primary_table = results.loc[ results['minimum_wells_threshold'] == primary ].copy() called = primary_table['significant'].astype(bool) wide_enough = primary_table['passes_effect_size'].astype(bool) significant = primary_table.loc[called & wide_enough].copy() if effect_threshold is not None: print(f"Effect-size cut removed {int((called & ~wide_enough).sum())} " f"of {int(called.sum())} guides that passed correction but " f"whose effect is narrower than {float(effect_threshold):.3g}.") wanted_level = str(settings.get('level') or 'both').strip().lower() wants_gene = wanted_level in ('gene', 'both') if 'guide_permutation_gene_level' in settings: wants_gene = bool(settings.get('guide_permutation_gene_level')) gene_primary = None if wants_gene: try: from .guide_permutation import analyse_long_gene_table gene_results = analyse_long_gene_table( data, outcomes, min_wells=thresholds, block_column=str(settings.get('guide_permutation_block', 'plateID')), nuisance_columns=list(settings.get('guide_nuisance_columns') or []), n_permutations=int(settings.get('guide_permutations', 200000)), random_state=int(settings.get('guide_permutation_seed', 0)), multiple_testing=str(settings.get('multiple_testing_method', 'fdr_bh')), alpha=float(settings.get('fdr_alpha', 0.05)), presence_threshold=float( settings.get('guide_presence_threshold', 0.0)), batch_size=int(settings.get('guide_permutation_batch_size', 500)), ) gene_results['feature'] = ( 'gene_fraction:gene[' + gene_results['gene'].astype(str) + ']') gene_results['grna'] = None gene_results['coefficient'] = gene_results[ 'standardized_marginal_effect'] gene_results['p_value'] = gene_results['permutation_p_value'] gene_results['q_value'] = gene_results['adjusted_p_value'] gene_results['condition'] = label_control_condition( gene_results['feature'], gene_results['gene'], nc=settings.get('negative_control_id'), pc=settings.get('positive_control_id'), controls=settings.get('nontargeting_control_grnas')) gene_primary = gene_results.loc[ gene_results['minimum_wells_threshold'] == primary].copy() print(f"Gene pass: {len(gene_primary)} genes tested as sets in " f"the primary >={primary}-well family, corrected as their " f"OWN BH family beside the {len(primary_table)} guides.") except Exception as error: # noqa: BLE001 - the guide pass still stands print(f"The gene-level permutation pass could not run: " f"{type(error).__name__}: {error}. results_gene.csv will be " f"empty; the guide results are unaffected.") gene_primary = None compatibility = { 'results': os.path.join(destination, 'results.csv'), 'results_grna': os.path.join(destination, 'results_grna.csv'), 'results_gene': os.path.join(destination, 'results_gene.csv'), 'significant': os.path.join(destination, 'results_significant.csv'), } levelled = primary_table.copy() levelled['level'] = 'grna' gene_rows = None if gene_primary is not None and len(gene_primary): gene_rows = gene_primary.copy() if wanted_level == 'gene' and gene_rows is not None: combined = gene_rows elif gene_rows is not None and wanted_level != 'grna': combined = pd.concat([levelled, gene_rows], ignore_index=True, sort=False) else: combined = levelled combined.to_csv(compatibility['results'], index=False) primary_table.to_csv(compatibility['results_grna'], index=False) (gene_primary if gene_primary is not None else primary_table.iloc[0:0]).to_csv( compatibility['results_gene'], index=False) significant.to_csv(compatibility['significant'], index=False) paths.update(compatibility) return { 'analysis_mode': 'guide_permutation', 'results': combined, 'families': results, 'gene_results': gene_primary, 'primary': primary_table, 'significant': significant, 'primary_min_wells': primary, 'effect_size_threshold': effect_threshold, 'effect_size_rule': effect_rule, 'paths': {key: str(path) for key, path in paths.items()}, } #: Settings a run chose for itself because the user left them unset. Filled #: by :func:`perform_regression` as each is derived, and printed once both #: are known -- the settings table is rendered before either exists. _AUTOMATIC_SETTINGS: dict = {} def _perform_regression_set_paths(settings): """Resolve and reserve a run's result directory and output paths. :param settings: Normalized regression settings; updated with resolved ``src`` and ``_regression_folder``. :returns: Results, gene, guide, and significant CSV paths, followed by the results directory and first count-data path. """ csv_path = settings['count_data'][0] from . import tabular remote_count = tabular._backend_of(csv_path) == 'postgres' automatic = '' if remote_count else os.path.dirname(csv_path) requested = settings.get('src') blank = str(requested or '').strip() in ('', 'path', '/path', '/path/to/src') src, how = ('', '') if remote_count and ( blank or tabular._backend_of(requested) == 'postgres') else \ resolve_regression_src(requested, automatic) if remote_count and not src: from .qt.i18n import tr raise ValueError(tr( 'A local regression output directory is required for PostgreSQL counts.')) settings['src'] = src if how != 'automatic': print(how) kind = results_folder_kind(settings) res_folder = _next_results_folder(os.path.join(src, 'results'), kind) _stage(settings, "placing the results folder") settings["_regression_folder"] = res_folder os.makedirs(res_folder, exist_ok=True) results_filename = 'results.csv' results_filename_gene = 'results_gene.csv' results_filename_grna = 'results_grna.csv' hits_filename = 'results_significant.csv' results_path=os.path.join(res_folder, results_filename) results_path_gene=os.path.join(res_folder, results_filename_gene) results_path_grna=os.path.join(res_folder, results_filename_grna) hits_path=os.path.join(res_folder, hits_filename) return results_path, results_path_gene, results_path_grna, hits_path, res_folder, csv_path
[docs] def results_folder_kind(settings) -> str: """What a run's results folder is NAMED after. The inference method when it decides the answer, and the regression type otherwise. Under `analysis_mode='guide_permutation'` the regression type is never read -- ols and mixed produce byte-identical results -- so a folder called `ridge` would name something the run did not do. PUBLIC, AND THE ONLY COPY. A test that re-derived this rule went stale when the rule changed and reported 39 missing CSVs while every run that wrote them was fine, which is the failure the `results_dir` helper in tests/test_cov_ml_perform_regression.py was already written to prevent once. A suite pointing at the wrong file is worse than a silent one. :param settings: run settings mapping, or ``None`` (treated as empty); only ``analysis_mode`` and ``regression_type`` are read. :returns: ``'guide_permutation'``, ``'auto'`` when no regression type is set, or the regression type as a string. """ settings = settings or {} if settings.get('analysis_mode') == 'guide_permutation': return 'guide_permutation' if settings.get('regression_type') is None: return 'auto' return str(settings['regression_type'])
def _next_results_folder(root, kind, limit=1000): """``<root>/<kind>``, or ``<kind>_1``, ``<kind>_2`` ... if taken. A run never writes on top of an earlier one. The old fixed path meant comparing two corrections, or re-running with one setting changed, left only the last on disk with nothing said about it -- and the results the user was looking at were not the results they thought. A folder counts as taken when it EXISTS AND HAS ANYTHING IN IT. An empty one is a directory somebody made and did not fill, and stepping past it would strand it forever. :param limit: stop after this many, rather than spinning if a filesystem keeps answering "yes, that exists too". """ import os base = os.path.join(root, str(kind)) for index in range(limit): candidate = base if index == 0 else f"{base}_{index}" try: if not os.path.isdir(candidate) or not os.listdir(candidate): return candidate except OSError: continue return f"{base}_{limit}" def _bracketed_identifier(pattern, text): """The id inside ``pattern``'s bracket, or ``None`` when there is none. The inline version of this was ``re.search(...).group(1) if 'grna' in x else None``, which assumes a term containing the word also contains the bracket. It does not: a mixed fit's variance component is named ``'grna Var'``, so the search returns None and ``.group(1)`` raises ``AttributeError`` on a run that had already fitted its model. """ match = re.search(pattern, str(text)) return match.group(1) if match else None def _annotate_level_coefficients(coef_df, n_grna, n_gene): """Attach the guide / gene id and the per-id row counts to ONE fit's table. :param coef_df: one level's coefficient table, straight out of :func:`regression`. :param n_grna: value_counts frame, one row per guide. :param n_gene: value_counts frame, one row per gene. :returns: a new frame with ``grna``, ``gene``, ``n_grna`` and ``n_gene``. """ coef_df = coef_df.copy() coef_df['grna'] = coef_df['feature'].map( lambda value: _bracketed_identifier(r'grna\[(.*?)\]', value)) coef_df['gene'] = coef_df['feature'].map( lambda value: _bracketed_identifier(r'gene\[(.*?)\]', value)) carried = dict(getattr(coef_df, "attrs", {}) or {}) coef_df = coef_df.merge(n_grna, how='left', on='grna', validate='many_to_one') coef_df = coef_df.merge(n_gene, how='left', on='gene', validate='many_to_one') if carried: coef_df.attrs.update(carried) return coef_df def _level_control_rows(frame, level, controls): """The control rows of ONE fit's table, matched at that fit's own level. ``settings['nontargeting_control_grnas']`` names GUIDES. The guide fit matches them whole, exactly as it always has. The gene fit has no guide column at all -- every ``gene_fraction:gene[...]`` row carries ``grna=None`` -- so matching the same list there selects nothing and the gene table silently gets no effect-size cut. A control guide identifies its gene by spaCR's own rule (:func:`spacr.hits.gene_of`: truncate at the first underscore), so the gene fit matches on that. """ if not (controls or []): return frame.iloc[0:0] from .control_names import matches, resolve_controls guides = frame['grna'] if 'grna' in frame.columns else frame.index library = [str(g) for g in pd.Series(guides).astype(str).unique()] genes = frame['gene'] if 'gene' in frame.columns else None specs = resolve_controls(controls, names=library) if not specs: return frame.iloc[0:0] keep = None for spec in specs: if level == 'gene': from .control_names import GENE, ControlSpec spec = ControlSpec(spec.typed, GENE, spec.value.split('_')[0] if not spec.is_gene else spec.value, spec.prefix) mask = matches(spec, frame['gene'].astype(str), frame['gene'].astype(str)) else: mask = matches(spec, pd.Series(guides).astype(str), genes) keep = mask if keep is None else (keep | mask) return frame.loc[keep.to_numpy()] #: What `annotation_source` calls the bundled, offline Toxoplasma path. BUNDLED_ANNOTATION = "toxoplasma" def _annotation_source(settings) -> str: """Which organism's annotation this run asked for, or "" for none. `annotation_source` is the one setting that says it. A dict that still carries the retired `Toxoplasma` or `toxo` switch is read through :func:`spacr.settings._fold_toxoplasma` on a copy, so a caller that hands this module a raw, unfolded dict gets the same answer the regression defaults would give it: a name wins, and otherwise true means the bundled tables and false means no annotation. """ from .settings import _fold_toxoplasma folded = _fold_toxoplasma(dict(settings or {}), quiet=True) return str(folded.get('annotation_source', '') or '').strip() def _toxoplasma_is_on(settings) -> bool: """Whether this run's annotation is the bundled *Toxoplasma* one. It gates the Toxoplasma-only figures -- the hyperLOPIT volcano and the GT1/ME49 phenotype and expression reports -- which mean nothing on another organism's screen. Before 2026-09-19 it read the `Toxoplasma` switch, which defaulted on, so a run annotated with 'human' still drew them. The name in `annotation_source` decides now, through the same resolver the annotation itself uses. """ source = _annotation_source(settings) if not source: return False from .uniprot import resolve return resolve(source).kind == "bundled" def _annotation_cache(settings): """Where a UniProt answer is kept, so a rerun needs no network.""" src = settings.get('src') if isinstance(src, (list, tuple)): src = src[0] if src else None if not src: return None return os.path.join(str(src), 'annotation_cache') def _call_level_hits(coef_df, level, settings, regression_type, merged_df, dependent_variable, bootstrap=None): """Correct one fit within itself and call that fit's hits. Treat guide and gene fits as separate multiple-testing families. They use the same wells and gene regressors are sums of guide regressors, so pooling both levels would count correlated hypotheses as independent tests. :param coef_df: one fit's annotated coefficient table. :param level: ``'grna'`` or ``'gene'`` -- which fit this is. :param regression_type: the backend that produced it. :param merged_df: the pre-clean long frame, for the lasso bootstrap. :param bootstrap: ``perform_regression``'s ``bootstrap_selection_frequencies`` closure. It is defined inside that function, so it cannot be looked up from here and the penalised backends need it passed in. :returns: ``(coef_df, significant, reg_threshold, effect_rule)``. """ from .thresholds import coefficient_threshold coef_df = coef_df.copy() reg_threshold = 0 effect_rule = 'no effect-size cut' if settings['nontargeting_control_grnas'] is not None: control_coef_df = _level_control_rows( coef_df, level, settings['nontargeting_control_grnas']) measured_threshold, threshold_rule = coefficient_threshold( control_coef_df['coefficient'], method=settings['threshold_method'], multiplier=settings['threshold_multiplier'], centre=None) effect_rule = threshold_rule print(f"Effect-size cut ({level}): {threshold_rule}") reg_threshold = (0 if measured_threshold is None else float(measured_threshold)) else: print(f"Effect-size cut ({level}): no control gRNAs were named, so " f"there is none; a hit is the corrected P value alone.") if regression_type in NO_P_VALUE_TYPES and bootstrap is None: raise ValueError( f"regression_type={regression_type!r} ranks features by bootstrap " f"selection frequency and has no p-value to correct, so " f"_call_level_hits needs perform_regression's " f"bootstrap_selection_frequencies passed as `bootstrap`.") if regression_type in NO_P_VALUE_TYPES: n_boot = settings.get('lasso_n_boot', 200) sel_threshold = settings.get('lasso_selection_threshold', 0.6) formula = prepare_formula( dependent_variable, random_row_column_effects=False, block_screen=screen_is_blockable(merged_df), level=level, model_plate_position=settings.get('model_plate_position', True)) cleaned_df = check_and_clean_data(merged_df.copy(), dependent_variable) sel_df = bootstrap( X=cleaned_df, y=cleaned_df[dependent_variable], formula=formula, alpha=settings.get('alpha', 'auto'), n_boot=n_boot, random_state=0, regression_type=regression_type, l1_ratio=settings['l1_ratio'], group_lasso_lambda=settings.get('group_lasso_lambda', 'auto'), ) coef_df = coef_df.merge(sel_df, on='feature', how='left', validate='one_to_one') significant = coef_df[ (coef_df['coefficient'] != 0) & (coef_df['selection_frequency'] >= sel_threshold) ].copy() significant = significant.sort_values( by='coefficient', key=lambda c: c.abs(), ascending=False, ) significant = significant[~significant['feature'].str.contains('row|column')] return coef_df, significant, reg_threshold, effect_rule from .multiple_testing import adjust_p_values, canonical_method method = canonical_method(settings.get('multiple_testing_method', 'fdr_bh')) alpha = float(settings.get('fdr_alpha', 0.05)) cut_alpha = float(settings.get('p_threshold_alpha', alpha) or alpha) cut_kind = str(settings.get('p_threshold_kind', 'adjusted')).strip().lower() cut_column = 'p_value' if cut_kind == 'raw' else 'q_value' from .hits import tested_family tested = pd.Series(tested_family(coef_df['feature']), index=coef_df.index) tested &= coef_df['p_value'].notna() if 'term_type' in coef_df.columns: tested &= coef_df['term_type'].eq(TERM_FIXED) coef_df['q_value'] = np.nan coef_df['multiple_testing_method'] = method if tested.any(): adjusted, _rejected = adjust_p_values( coef_df.loc[tested, 'p_value'].to_numpy(dtype=float), method=method, alpha=alpha) coef_df.loc[tested, 'q_value'] = adjusted raw_hits = int((coef_df.loc[tested, 'p_value'] <= alpha).sum()) corrected_hits = int((coef_df.loc[tested, 'q_value'] < alpha).sum()) print(f"Multiple testing ({level}): {method} across {int(tested.sum())} " f"tested coefficients at alpha={alpha:g} — {raw_hits} pass the raw " f"P value, {corrected_hits} pass correction.") if cut_kind == 'raw' or cut_alpha != alpha: print(f" Calling hits on the {cut_kind} P at {cut_alpha:g}" + (", NOT corrected for multiple testing." if cut_kind == 'raw' else ".")) significant = coef_df.loc[coef_df[cut_column] < cut_alpha].copy() coef_df['effect_size_threshold'] = ( np.nan if not reg_threshold else abs(float(reg_threshold))) coef_df['effect_size_rule'] = effect_rule significant = significant.assign( effect_size_threshold=(np.nan if not reg_threshold else abs(float(reg_threshold))), effect_size_rule=effect_rule) if reg_threshold: wide_enough = (significant['coefficient'].abs() >= abs(reg_threshold)) called = len(significant) significant = significant.loc[wide_enough].copy() print(f"Effect-size cut ({level}) removed {called - len(significant)} " f"of {called} coefficients that passed correction but whose " f"effect is narrower than {abs(reg_threshold):.3g}.") significant = significant.sort_values( by='coefficient', ascending=False) significant = significant[~significant['feature'].str.contains('row|column')] return coef_df, significant, reg_threshold, effect_rule def _stage(settings, name): """Record the current fit stage, announce it, and never raise. Fall back to storing ``_regression_stage`` in the settings mapping when resource measurement is unavailable. THE ANNOUNCEMENT IS THE POINT AS MUCH AS THE RECORD. Reading the counts and fitting the model each take minutes on a four-plate screen, and a step that prints nothing while it runs cannot be told apart from a step that has hung -- which is how a working run comes to be reported as a dead one. The recorded resident size goes on the same line, because the other thing a long silent step invites is a guess about memory. """ reading = {} try: from .fit_resources import record_stage reading = record_stage(settings, name) except Exception: # noqa: BLE001 try: settings["_regression_stage"] = str(name) except Exception: # noqa: BLE001 pass try: rss = reading.get("rss") if isinstance(reading, dict) else None note = f" (resident {rss / 1e9:.1f} GB)" if rss else "" print(f"Regression: {name}{note}.", flush=True) except Exception: # noqa: BLE001 pass return reading #: Panels that need a fitted object exposing residuals, and the models that #: cannot supply one. RRA is a rank statistic -- it never fits a linear #: predictor, so "residual" has no meaning for it rather than being #: unavailable. Naming them here rather than catching AttributeError keeps the #: REASON: a missing QQ plot and an inapplicable one look identical to a #: reader, and only the inapplicable case is valid. RESIDUAL_FREE_MODELS: dict = { "rra": ("Robust Rank Aggregation is a rank statistic: it ranks guides " "within each well and aggregates those ranks, so it never forms a " "linear predictor and there is no residual to plot."), "horseshoe": ("The horseshoe fit is sampled rather than solved, so it has " "a posterior rather than one set of fitted values."), } def _diagnostic_inputs(model): """``(observed, fitted, design)`` from a fitted model, or ``(None,)*3``. Duck-typed on purpose. statsmodels results expose ``fittedvalues``, ``resid`` and ``model.exog``; the backends spaCR wraps do not share a base class, so asking what an object HAS is the only question that works across all of them. """ fitted = getattr(model, "fittedvalues", None) resid = getattr(model, "resid", None) if fitted is None or resid is None: return None, None, None try: observed = np.asarray(fitted, dtype=float) + np.asarray(resid, dtype=float) except Exception: # noqa: BLE001 return None, None, None design = getattr(getattr(model, "model", None), "exog", None) return observed, np.asarray(fitted, dtype=float), design def _diagnostic_screen_design(data, settings): """Return the well-by-guide matrix and aligned block labels for QC. ``perform_regression`` works with a long table because the historical OLS formula has one fitted row per well-guide pair. Identifiability, however, is a question about independent wells and guide predictors. Passing that long mixed-type table straight to ``design_report`` both miscounted the observations and failed while converting ``prc`` to float. This adapter performs the same sum-and-zero-fill pivot used by the permutation path and verifies that a well has exactly one block label. A caller that already supplies a numeric wide matrix keeps the original API: it is returned unchanged with no inferred block. """ if not isinstance(data, pd.DataFrame): return data, None required = {schema.PRC_KEY, "grna", "fraction"} if not required.issubset(data.columns): return data, None frame = data.loc[:, [schema.PRC_KEY, "grna", "fraction"]].copy() if frame[[schema.PRC_KEY, "grna"]].isna().any().any(): raise ValueError("well and guide identifiers must not contain missing values") frame["fraction"] = pd.to_numeric(frame["fraction"], errors="raise") values = frame["fraction"].to_numpy(dtype=float) if not np.isfinite(values).all(): raise ValueError("guide fractions must be finite") if np.any(values < 0): raise ValueError("guide fractions must be non-negative") wide = frame.pivot_table( index=schema.PRC_KEY, columns="grna", values="fraction", aggfunc="sum", fill_value=0.0, ).sort_index() wide.columns = wide.columns.astype(str) block = None block_column = str(settings.get("guide_permutation_block") or schema.PLATE_KEY) if block_column in data.columns: labels = data.loc[:, [schema.PRC_KEY, block_column]].copy() counts = labels.groupby(schema.PRC_KEY, sort=False)[block_column].nunique( dropna=False) inconsistent = counts > 1 if inconsistent.any(): example = inconsistent.index[inconsistent][0] raise ValueError( f"block labels are not constant within well {example!r}") block = (labels.drop_duplicates(schema.PRC_KEY) .set_index(schema.PRC_KEY)[block_column] .reindex(wide.index)) if block.isna().any(): raise ValueError("block labels are missing for one or more wells") return wide, block def _write_regression_diagnostics(res_folder, fractions, fits, settings): """Write the diagnostic suite for a completed fit. THE DESIGN REPORT IS UNCONDITIONAL. It needs no fit at all -- only the well-by-guide matrix -- so it is available for every model including RRA, and it is the one that would have caught the failure :mod:`spacr.regression_diagnostics` was written for: 824 guides in 587 wells returning a confident P value for every guide out of a rank-deficient matrix. RESIDUAL PANELS SAY WHY WHEN THEY CANNOT BE DRAWN. `write_diagnostic_suite` skips a block whose inputs are absent, silently, which is right for a library and wrong here: the user asked for these plots "whenever possible", and the interesting case is precisely when it is not possible. So a model that cannot support residuals writes a note naming the reason beside the panels that did run. Never raises. A diagnostic that took the analysis down with it would be worse than no diagnostic -- the numbers the user came for are already computed by the time this runs. """ from . import regression_diagnostics as rd if not res_folder: return {} destination = os.path.join(res_folder, "diagnostics") written: dict = {} try: model, _coef, model_type = next(iter(fits.values())) except Exception: # noqa: BLE001 model, model_type = None, str(settings.get("regression_type") or "") observed, fitted, design = _diagnostic_inputs(model) reason = RESIDUAL_FREE_MODELS.get(str(model_type).lower()) if observed is None and reason is None: reason = (f"The {model_type or 'selected'} backend did not expose " "fitted values and residuals, so the residual panels could " "not be computed for this run.") try: fractions, block = _diagnostic_screen_design(fractions, settings) written = dict(rd.write_diagnostic_suite( destination, fractions=fractions, block=block, observed=observed, fitted=fitted, design=design, label=str(model_type or ""), presence_threshold=float( settings.get("guide_presence_threshold", 0.0) or 0.0))) except Exception as error: # noqa: BLE001 print(f"Diagnostics could not be written: " f"{type(error).__name__}: {error}") return {} if observed is None and reason: note_path = os.path.join(destination, "residual_panels_not_available.txt") try: with open(note_path, "w", encoding="utf-8") as handle: handle.write(reason + "\n") written["residuals_unavailable"] = note_path except OSError: pass print(f"Residual diagnostics were not computed: {reason}") return written def _write_regression_panel_packages(outcome, settings): """Build requested publication panels from this run's final CSV files.""" manifest = settings.get("regression_panel_manifest") if manifest is None: return None settings["_regression_stage"] = "writing publication panel packages" if not isinstance(outcome, dict): raise TypeError("A regression panel manifest needs a mapping outcome") raw_folder = ( outcome.get("res_folder") or settings.get("_regression_folder") or "" ) if not raw_folder: raise ValueError("A regression panel manifest needs a results folder") res_folder = os.path.abspath(os.fspath(raw_folder)) settings["_regression_folder"] = res_folder paths = outcome.get("paths") paths = paths if isinstance(paths, dict) else {} final_paths = { "grna": os.path.abspath(os.fspath( paths.get("results_grna") or os.path.join(res_folder, "results_grna.csv") )), "gene": os.path.abspath(os.fspath( paths.get("results_gene") or os.path.join(res_folder, "results_gene.csv") )), } missing = [path for path in final_paths.values() if not os.path.isfile(path)] if missing: raise FileNotFoundError( "Regression panel source files do not exist: " + ", ".join(missing) ) configured = settings.get("dependent_variable") if isinstance(configured, str): phenotypes = [configured] elif isinstance(configured, (list, tuple)): phenotypes = list(configured) else: phenotypes = [] if not phenotypes or any(not isinstance(value, str) or not value.strip() for value in phenotypes): raise ValueError( "regression_panel_manifest needs one or more named " "dependent_variable values" ) phenotypes = [value.strip() for value in phenotypes] fdr_alpha = float(settings.get("fdr_alpha", 0.05)) artifacts = {} for level, path in final_paths.items(): table = tabular.read_table(path, report=None) for phenotype in phenotypes: selected = table if len(phenotypes) > 1: if "outcome" not in table.columns: raise ValueError( f"Multi-phenotype result {path} has no 'outcome' column" ) selected = table.loc[table["outcome"].eq(phenotype)].copy() if selected.empty: raise ValueError( f"Result {path} has no rows for phenotype {phenotype!r}" ) artifacts[f"{phenotype}_{level}"] = { "phenotype": phenotype, "level": level, "data": selected, "run_artifact_path": path, "res_folder": res_folder, "dependent_variable": phenotype, "fdr_alpha": fdr_alpha, } from .regression_panels import build_manifest_packages return build_manifest_packages( manifest, artifacts, os.path.join(res_folder, "publication_panels"), )
[docs] def perform_regression(settings): """Run the regression and report actionable details if it fails. On failure, the original exception is re-raised unchanged after a report is printed and written to the run folder. The report includes the most recent stage stored in ``settings['_regression_stage']``, available design dimensions, and a remedy for recognized failures. :param settings: Regression settings consumed by the fitting pipeline. :returns: Regression output mapping. ``model_data`` is the prepared input table, not coefficient results; ``fit_designs`` records measured design counts separately for each parametric fit level. Unrecorded counts are omitted, and permutation outputs retain their own result schema. :raises Exception: Re-raises the original regression failure. """ from .regression_failure import describe_failure, write_failure_report try: settings.setdefault("_regression_stage", "starting") except AttributeError: pass try: outcome = _perform_regression(settings) publication_panels = _write_regression_panel_packages(outcome, settings) if publication_panels is not None: outcome["publication_panels"] = publication_panels except Exception as error: # noqa: BLE001 stage = "" folder = "" frame = None try: stage = str(settings.get("_regression_stage", "") or "") folder = str(settings.get("_regression_folder", "") or "") frame = settings.get("_regression_frame") except AttributeError: pass print(describe_failure(error, stage=stage, settings=settings, frame=frame, include_traceback=False)) written = write_failure_report(folder, error, stage=stage, settings=settings, frame=frame) if written: print(f"The full report, with the traceback, is in {written}") raise _write_fit_resources(outcome, settings) return outcome
#: What a completed run's resource record is called, beside its results. FIT_RESOURCES_FILENAME = "fit_resources.txt" def _write_fit_resources(outcome, settings): """Write per-stage and peak resource use for a successful fit. Store the report beside the run results when a destination and measurements are available. Return an empty string on missing data or any measurement/write failure so resource reporting cannot fail the fit. """ try: from .fit_resources import describe_resources, peak folder = "" if isinstance(outcome, dict): folder = str(outcome.get("res_folder") or "") if not folder: folder = str(settings.get("_regression_folder", "") or "") table = describe_resources(settings) if not folder or not table or not os.path.isdir(folder): return "" high = peak(settings) lines = ["WHAT THIS FIT COST", "==================", "", "Recorded per stage as the run went. 'not measured' is not " "zero:", "psutil absent, or no CUDA tensor allocated yet.", "", table, ""] if not high: lines.append("No reading could be taken on this machine.") path = os.path.join(folder, FIT_RESOURCES_FILENAME) with open(path, "w", encoding="utf-8") as handle: handle.write("\n".join(lines) + "\n") return path except Exception: # noqa: BLE001 return "" def _warn_if_penalised_no_hits(settings, coef_df): """Explain why a penalised fit with no small P values is inconclusive.""" penalised = str(settings.get('regression_type', '')).lower() in ( 'ridge', 'lasso', 'elasticnet') if penalised and len(coef_df): p_values = pd.to_numeric(coef_df.get('p_value'), errors='coerce') if not (p_values < 0.05).any(): print( f"\nNOTE: {settings['regression_type']} returned no " f"coefficient below p=0.05. Its p-values are conservative " f"by construction -- the standard error is unpenalised " f"while the coefficient it is divided into has been shrunk " f"-- so this is NOT evidence of no effect. Refit with " f"regression_type='ols' (or 'rlm' for a robust check) " f"before concluding anything from it.") return True return False def _perform_regression(settings): """Regress per-well phenotype scores against gRNA / gene counts to identify hits from a pooled CRISPR screen. Reads one or more score CSVs (from :func:`generate_ml_scores` or a deep-learning classifier) and one or more sgRNA count CSVs (from :func:`spacr.sequencing.generate_barecode_mapping`), aligns them on plate / well, fits the requested regression model, merges metadata, and emits volcano plots, plate heatmaps, gene phenotype plots and GO enrichment reports. :param settings: Settings dict, canonicalized via :func:`spacr.settings.get_perform_regression_default_settings`. Key entries: - ``paired_data`` — ordered rows that explicitly pair one score CSV with one sgRNA-count CSV. Plate identity comes from the files when they agree, from the partner when only one declares it, or from the pair-row order when neither does. Legacy ``score_data`` and ``count_data`` lists are migrated positionally with a visible log. - ``dependent_variable`` — column of ``score_data`` to regress (e.g. ``'pred'``, ``'recruitment'``, ``'pathogen_nucleus_shortest_distance'``). - ``regression_type`` — any name in :data:`REGRESSION_TYPES`, or ``None`` to choose one from the response distribution. See :func:`regression_model` for what each backend is for. - ``analysis_mode='guide_permutation'`` — instead test plate-adjusted marginal guide associations with empirical P values and apply ``multiple_testing_method`` (Benjamini--Hochberg by default) within each requested ``guide_min_wells`` family. - the per-model settings each backend reads — ``alpha``, ``l1_ratio``, ``cov_type``, ``quantile``, ``hinge_threshold``, ``hinge_n_boot``, ``huber_t``, ``random_row_column_effects``. A setting the chosen type cannot read is refused rather than ignored; see :data:`REGRESSION_SETTINGS_USED`. - ``batch_correction`` — optional ``combat``, ``center``, ``zscore``, ``robust_zscore`` or reference-control ``control_center`` normalization of the dependent variable before well aggregation. - ``fraction_threshold``, ``min_observations_per_hit``, ``metadata_files``, ``volcano``, ``heatmap_feature``. :returns: Path to the merged, metadata-annotated results DataFrame (also written to ``results/<score_source>/<regression_type>/ results.csv``). Related gene/gRNA CSVs and significance calls are saved alongside. :raises ValueError: if paired files declare incompatible plate IDs, ``dependent_variable`` is not a score column, ``regression_type`` is unsupported, or a guide-permutation support family or correction setting is invalid. Example: .. code-block:: python from spacr.ml import perform_regression settings = { 'paired_data': [{ 'score': '/data/plate01/results/xgb_scores.csv', 'count': '/data/plate01/sequencing/counts.csv', }], 'dependent_variable': 'pred', 'regression_type': 'mixed', } perform_regression(settings) See Also: :func:`generate_ml_scores` — produce the ``score_data`` input. :func:`spacr.sequencing.generate_barecode_mapping` — produce the ``count_data`` input. """ from .plot import plot_plates, plot_data_from_csv from .utils import merge_regression_res_with_metadata, save_settings, correct_metadata from .settings import get_perform_regression_default_settings from .toxo import custom_volcano_plot, plot_gene_phenotypes, plot_gene_heatmaps def _perform_regression_read_data(settings): """Load paired inputs, validate the analysis, and return both frames.""" _stage(settings, "reading the input tables") pairs, _migrated = normalize_regression_input_pairs(settings) count_data_df, score_data_df, audit = \ load_regression_input_pairs(pairs) settings['paired_data'] = pairs settings['input_pair_audit'] = audit print(f"Score data: {len(score_data_df)} rows from " f"{len(settings['score_data'])} file(s)") print(f"Count data: {len(count_data_df)} rows from " f"{len(settings['count_data'])} file(s)") print(f"Dependent variable: {len(score_data_df)}") print(f"Independent variable: {len(count_data_df)}") if settings['dependent_variable'] not in score_data_df.columns: if not settings['dependent_variable'] == 'pathogen_nucleus_shortest_distance': looks_like_counts = sorted( {'grna', 'grna_name', 'count'}.intersection( score_data_df.columns)) if looks_like_counts: hint = ( f"\n\nThe score table has {looks_like_counts} and " f"no score column, which is the shape of a COUNT " f"file. The score and count inputs look swapped.") else: numeric = [ column for column in score_data_df.columns if pd.api.types.is_numeric_dtype( score_data_df[column]) and column not in {'plateID', 'rowID', 'columnID', 'fieldID', 'objectID', 'count'}] hint = (f"\n\nColumns that could be the response: " f"{numeric[:12]}") if numeric else "" raise ValueError( f"dependent_variable=" f"{settings['dependent_variable']!r} is not a column " f"of the score table, which has " f"{list(score_data_df.columns)[:15]}.{hint}") _reject_impossible_probabilities(settings) mode = str(settings.get('analysis_mode', 'regression')).strip().lower() if mode not in {'regression', 'guide_permutation'}: raise ValueError( f"Unsupported analysis_mode {mode!r}; choose 'regression' " "or 'guide_permutation'.") settings['analysis_mode'] = mode if mode == 'regression': reg_type = settings['regression_type'] if reg_type is not None and reg_type not in REGRESSION_TYPES: if reg_type in UNSUPPORTED_REGRESSION_TYPES: raise ValueError( f"Unsupported regression type {reg_type}: " f"{UNSUPPORTED_REGRESSION_TYPES[reg_type]}") print(f'Possible regression types: ' f'{list(REGRESSION_TYPES) + [None]}') raise ValueError(f"Unsupported regression type {reg_type}") _reconcile_random_row_column_effects(settings) _reject_unused_run_settings(settings) return count_data_df, score_data_df def _count_variable_instances(df, column_1, column_2): """Return ``df`` and value-count tables for both named columns.""" for col in (column_1, column_2): if col not in df.columns: raise KeyError( f"Column '{col}' not found in independent_df. " f"Available columns: {list(df.columns)}" ) n_grna = df[column_1].value_counts().reset_index() n_grna.columns = [column_1, f"n_{column_1}"] n_gene = df[column_2].value_counts().reset_index() n_gene.columns = [column_2, f"n_{column_2}"] return df, n_grna, n_gene def _qc_plot(plot_settings): """Render one QC plot, reporting - not raising - on failure. The QC tables written between these calls are data outputs, so a plotting failure must not cost them. spacrGraph runs a group comparison, which scipy rejects with "Must enter at least two input sample vectors" on the very common single-plate run. """ try: return plot_data_from_csv(settings=plot_settings) except Exception as e: print(f"Skipping QC plot {plot_settings['graph_name']!r}: {e}") return None, None def grna_metricks(df): """Return per-gRNA and per-well coverage counts derived from a long ``prc`` DataFrame. :param df: DataFrame with ``prc``, ``grna`` and ``gene`` columns. :returns: ``(final_grna_df, prc_gene_count_df)`` — per-gRNA well counts and per-well distinct-gene counts. """ _assign_prc_parts(df) grna_well_counts = (df.groupby(['grna', 'plateID'])['prc'].nunique().reset_index(name='grna_well_count')) gene_well_counts = (df.groupby(['gene', 'plateID'])['prc'].nunique().reset_index(name='gene_well_count')) unique_triplets = df[['grna', 'gene', 'plateID']].drop_duplicates() merged_df = pd.merge(unique_triplets, grna_well_counts, on=['grna', 'plateID'], how='left', validate='many_to_one') merged_df = pd.merge(merged_df, gene_well_counts, on=['gene', 'plateID'], how='left', validate='many_to_one') final_grna_df = merged_df[['grna', 'plateID', 'grna_well_count', 'gene_well_count']] prc_gene_count_df = (df.groupby('prc')['gene'].nunique().reset_index(name='gene_count')) _assign_prc_parts(prc_gene_count_df) return final_grna_df, prc_gene_count_df def get_outlier_reference_values(df, outlier_col, return_col): """Return unique ``return_col`` values whose ``outlier_col`` falls outside 1.5*IQR. :param df: Input DataFrame. :param outlier_col: Numeric column screened for outliers. :param return_col: Column whose distinct values are returned. :returns: List of unique reference values for outlier rows. """ Q1 = df[outlier_col].quantile(0.05) Q3 = df[outlier_col].quantile(0.95) IQR = Q3 - Q1 lower_bound = Q1 - 1.5 * IQR upper_bound = Q3 + 1.5 * IQR outlier_mask = (df[outlier_col] < lower_bound) | (df[outlier_col] > upper_bound) outliers = df.loc[outlier_mask, return_col] outliers_ls = outliers.unique().tolist() return outliers_ls def bootstrap_selection_frequencies(X, y, formula, alpha='auto', n_boot=200, random_state=None, regression_type='lasso', l1_ratio=0.5, group_lasso_lambda='auto'): """Return per-feature selection frequencies from a nonparametric bootstrap. Output ranks features by how often their coefficient is non-zero across resamples; this is a stability score, not a hypothesis test. :param X: Long-form DataFrame; design matrix is built per resample from ``formula`` for stable factor levels. :param y: Response array aligned with ``X`` by index. :param formula: Patsy formula for ``dmatrices``. :param alpha: Regularisation strength; ``'auto'``/``None`` runs the cross-validated estimator per resample. :param n_boot: Number of bootstrap resamples. Default ``200``. :param random_state: Seed for the resampling RNG. :param regression_type: ``'lasso'``, ``'elasticnet'`` or ``'group_lasso'`` - the same penalty the reported coefficients were fitted with, or the frequencies would describe a different model from the one in ``results.csv``. :param l1_ratio: ``elasticnet`` mix; ignored for ``'lasso'``. :param group_lasso_lambda: the block penalty, for ``'group_lasso'``. It is a separate argument from ``alpha`` because it is a separate setting: ``alpha`` is not read by that backend at all, so a resample fitted at ``alpha`` would be a different model from the one whose coefficients this is ranking. :returns: DataFrame with columns ``feature``, ``selection_frequency`` and ``mean_coefficient``. :raises RuntimeError: if every resample fails to fit. """ rng = np.random.default_rng(random_state) n = len(X) use_cv = alpha is None or (isinstance(alpha, str) and alpha == 'auto') def _estimator(): """Return the configured sklearn lasso or elastic-net estimator.""" if regression_type == 'elasticnet': return (ElasticNetCV(l1_ratio=l1_ratio, cv=5, max_iter=10000) if use_cv else ElasticNet(alpha=alpha, l1_ratio=l1_ratio, max_iter=10000)) return (LassoCV(cv=5, max_iter=10000) if use_cv else Lasso(alpha=alpha, max_iter=10000)) y0, X0 = dmatrices(formula, data=X, return_type='dataframe') feature_index = pd.Index(X0.columns) blocks = (_design_column_groups(feature_index) if regression_type == 'group_lasso' else None) block_penalty = None if blocks is not None: from . import group_lasso as group_lasso_module if _left_blank(group_lasso_lambda) or ( isinstance(group_lasso_lambda, str) and group_lasso_lambda.strip().lower() == 'auto'): block_penalty = group_lasso_module.choose_lambda( np.asarray(X0, dtype=float), np.asarray(y0, dtype=float).ravel(), blocks) else: block_penalty = float(group_lasso_lambda) def _resample_coefficients(design, response): """Return group-lasso or sklearn coefficients for one resample.""" if blocks is not None: from . import group_lasso as group_lasso_module beta, _intercept, _converged = group_lasso_module.fit( np.asarray(design, dtype=float), np.asarray(response, dtype=float).ravel(), blocks, lam=block_penalty) return np.asarray(beta, dtype=float).ravel() return np.asarray(_estimator().fit(design, response).coef_).ravel() nonzero_counts = pd.Series(0.0, index=feature_index) coef_sums = pd.Series(0.0, index=feature_index) successful = 0 dropped = 0 last_failure = None for _ in range(n_boot): idx = rng.integers(0, n, size=n) boot = X.iloc[idx].reset_index(drop=True) try: yb, Xb = dmatrices(formula, data=boot, return_type='dataframe') except Exception as exc: dropped += 1 last_failure = exc continue Xb = Xb.reindex(columns=feature_index, fill_value=0.0) yb = np.asarray(yb).ravel() coefs = pd.Series(_resample_coefficients(Xb, yb), index=feature_index) nonzero_counts += (coefs != 0).astype(float) coef_sums += coefs successful += 1 if successful == 0: raise RuntimeError("All bootstrap resamples failed to fit. " "Check the formula and ensure factor levels are not too sparse.") if dropped: LOG.warning( "stability selection: %d of %d resamples produced no design " "matrix (last error: %s). selection_frequency and " "mean_coefficient below are over the remaining %d, not over " "%d.", dropped, n_boot, last_failure, successful, n_boot) return pd.DataFrame({ 'feature': feature_index, 'selection_frequency': (nonzero_counts / successful).values, 'mean_coefficient': (coef_sums / successful).values, }) settings = get_perform_regression_default_settings(settings) count_data_df, score_data_df = _perform_regression_read_data(settings) from .regression_layout import normalise_count_table_layout count_data_df, resolved_input_layout = normalise_count_table_layout( count_data_df, layout=settings.get('independent_variable_layout', 'auto'), guide_column=str(settings.get('count_grna_column') or 'grna'), count_column=str(settings.get('count_value_column') or 'count'), wide_predictor_columns=(settings.get('wide_predictor_columns') or None), ) settings['independent_variable_layout_resolved'] = resolved_input_layout print( f"Independent-variable input: {resolved_input_layout}; normalized to " f"long rows ({len(count_data_df):,} well-guide values)." ) if "rowID" in count_data_df.columns: count_data_df['rowID'] = ( count_data_df['rowID'].astype(str) .str.rsplit(schema.KEY_SEPARATOR, n=1).str[-1] ) if {'plateID', 'rowID', 'columnID'}.issubset(score_data_df.columns): score_data_df['prc'] = ( _compose_prc_column(score_data_df) ) if settings.get('verbose'): print("score_data_df plateID counts:") print(score_data_df['plateID'].value_counts()) print("count_data_df plateID counts:") print(count_data_df['plateID'].value_counts()) results_path, results_path_gene, results_path_grna, hits_path, res_folder, csv_path = _perform_regression_set_paths(settings) batch_method = str( settings.get('batch_correction', 'none') or 'none' ).strip().lower() if batch_method not in {'none', 'off', 'false'}: dependent_variable = settings['dependent_variable'] if dependent_variable not in score_data_df.columns: raise ValueError( f"Batch correction cannot run because dependent_variable=" f"{dependent_variable!r} is not present in the score table. " "Choose an existing score column or set batch_correction=none." ) from .batch_correction import ( correct_from_metadata, correction_kwargs, write_report, ) corrected, correction_report = correct_from_metadata( score_data_df[[dependent_variable]], score_data_df, batch_covariate_column=settings.get('batch_covariate_column'), batch_combat_mean_only=bool( settings.get('batch_combat_mean_only', False)), **correction_kwargs(settings), ) report_path = write_report( correction_report, os.path.join(res_folder, 'batch_correction.json'), ) shift = float(np.abs( np.asarray(corrected[dependent_variable], dtype=float) - np.asarray(score_data_df[dependent_variable], dtype=float) ).mean()) if len(score_data_df) else 0.0 n_batches = len(getattr(correction_report, 'batches', ()) or ()) print( f"Batch correction {correction_report.method}: " f"{correction_report.centroid_spread_before} -> " f"{correction_report.centroid_spread_after} centroid spread, " f"across {n_batches} batch(es); " f"{dependent_variable} moved by {shift:.6g} on average. " f"Report: {report_path}" ) if n_batches < 2: print( f" It changed nothing, and could not: batch correction " f"removes variance BETWEEN batches and this run has " f"{n_batches}. Set batch_column to a column that varies, or " f"batch_correction='none' -- the result is identical either " f"way." ) elif shift == 0.0: print( " It changed nothing: the batches already agree on " f"{dependent_variable}. That is a finding about the screen, " "not a failure of the correction." ) score_data_df.loc[:, dependent_variable] = corrected[ dependent_variable ] for note in correction_report.warnings: print(f"Warning: batch correction: {note}") save_settings(settings, name='regression', show=True) count_source = os.path.dirname(settings['count_data'][0]) volcano_path = os.path.join(res_folder, 'volcano_plot.pdf') if isinstance(settings['filter_value'], list): filter_value = list(settings['filter_value']) else: filter_value = [] try: from .well_spec import control_block_wells for well in control_block_wells(settings): if well not in filter_value: filter_value.append(well) except Exception: # noqa: BLE001 LOG.debug("could not resolve the control blocks", exc_info=True) filter_column = settings['filter_column'] score_data_df = clean_controls(score_data_df, settings['filter_value'], filter_column) try: from .outlier_filter import apply as _drop_outliers, describe score_data_df, _outlier_report = _drop_outliers(score_data_df, settings) _said = describe(_outlier_report) if _said: print(_said) except Exception as _error: # noqa: BLE001 print(f"[outliers] the pre-annotation filter did not run " f"({type(_error).__name__}: {_error}); the counts below are " f"unfiltered") if settings['verbose']: print(f"Dependent variable after clean_controls: {len(score_data_df)}") _AUTOMATIC_SETTINGS.clear() screen_folders = _screen_figure_folders(settings) sim_min_count = minimum_cell_simulation( settings, tolerance=settings['tolerance'], dst=res_folder) if settings['min_cells_per_well'] is None: settings['min_cells_per_well'] = sim_min_count _AUTOMATIC_SETTINGS['min_cells_per_well'] = sim_min_count if settings['verbose']: print(f"Minimum cell count: {settings['min_cells_per_well']}") print(f"Dependent variable after minimum cell count filter: {len(score_data_df)}") display(score_data_df) orig_dv = settings['dependent_variable'] _before_transform = None try: _before_transform, _ = process_scores( score_data_df, settings['dependent_variable'], None, settings['min_cells_per_well'], settings['agg_type'], None, settings['regression_type'], settings['invert_dependent_variable']) except Exception: # noqa: BLE001 _before_transform = None dependent_df, dependent_variable = process_scores( score_data_df, settings['dependent_variable'], None, settings['min_cells_per_well'], settings['agg_type'], settings['transform'], settings['regression_type'], settings['invert_dependent_variable']) _show_response_distribution(_before_transform, dependent_variable, settings) if settings['verbose']: print(f"Dependent variable after process_scores: {len(dependent_df)}") display(dependent_df) if settings.get('calibrate_fraction_threshold'): measured = _calibrated_fraction_threshold(settings) if measured is not None: settings['fraction_threshold'] = measured _AUTOMATIC_SETTINGS['fraction_threshold'] = measured if settings['fraction_threshold'] is None: before_sweep = _figure_stamps(screen_folders) settings['fraction_threshold'] = _graph_sequencing_stats(settings) _AUTOMATIC_SETTINGS['fraction_threshold'] = settings['fraction_threshold'] for kept in _keep_figures_with_the_run(before_sweep, screen_folders, res_folder): print(f"Kept with the run: {kept}") else: before_sweep = _figure_stamps(screen_folders) _draw_the_threshold_sweep( settings, res_folder, measured='fraction_threshold' in _AUTOMATIC_SETTINGS) for kept in _keep_figures_with_the_run(before_sweep, screen_folders, res_folder): print(f"Kept with the run: {kept}") if _AUTOMATIC_SETTINGS: print("\nChosen automatically (not set by the user):") for key, value in _AUTOMATIC_SETTINGS.items(): print(f" {key:<28}{value}") try: save_settings(settings, name='regression', show=False) except Exception as error: # noqa: BLE001 print(f"Could not re-save the resolved settings: {error}") _exclusions = settings.setdefault("_regression_exclusions", {}) _stage(settings, "reading the counts") _read_kwargs = { "filter_column": filter_column, "filter_value": filter_value, "record": _exclusions, } if settings.get('exclude_grnas'): _read_kwargs["exclude_grnas"] = settings['exclude_grnas'] independent_df = process_reads( count_data_df, settings['fraction_threshold'], None, **_read_kwargs) if settings['verbose']: print("independent_df columns:", list(independent_df.columns)) print("independent_df head:") print(independent_df.head()) print(independent_df) if settings['verbose']: print(f"Independent variable after process_reads: {len(independent_df)}") merge_validate = ( 'many_to_many' if settings['agg_type'] is None else 'many_to_one') merged_df = pd.merge(independent_df, dependent_df, on='prc', validate=merge_validate) _check_score_count_pairing(independent_df, dependent_df, merged_df, record=settings.get('_regression_exclusions')) _merged_for_counts, n_grna, n_gene = _count_variable_instances( merged_df, column_1='grna', column_2='gene') if settings['verbose']: display(independent_df) display(dependent_df) display(merged_df) _assign_prc_parts(merged_df) try: os.makedirs(res_folder, exist_ok=True) data_path = os.path.join(res_folder, 'regression_data.csv') merged_df.to_csv(data_path, index=False) print(f"Saved regression data to {data_path}") qc_graph_type = _qc_graph_type() cell_settings = {'src':data_path, 'graph_name':'cell_count', 'data_column':['cell_count'], 'grouping_column':'plateID', 'graph_type':qc_graph_type, 'theme':'bright', 'save':True, 'y_lim':[None,None], 'log_y':False, 'log_x':False, 'representation':'well', 'remove_outliers':False, 'verbose':False} _, _ = _qc_plot(cell_settings) final_grna_df, prc_gene_count_df = grna_metricks(merged_df) if settings['outlier_detection']: outliers_grna = get_outlier_reference_values(final_grna_df,outlier_col='grna_well_count',return_col='grna') if len (outliers_grna) > 0: merged_df = merged_df[ ~merged_df['grna'].isin(outliers_grna)].copy() final_grna_df, prc_gene_count_df = grna_metricks(merged_df) merged_df.to_csv(data_path, index=False) print(f"Saved regression data to {data_path}") grna_data_path = os.path.join(res_folder, 'grna_well.csv') final_grna_df.to_csv(grna_data_path, index=False) print(f"Saved grna per well data to {grna_data_path}") wells_per_gene_settings = {'src':grna_data_path, 'graph_name':'wells_per_gene', 'data_column':['grna_well_count'], 'grouping_column':'plateID', 'graph_type':qc_graph_type, 'theme':'bright', 'save':True, 'y_lim':[None,None], 'log_y':False, 'log_x':False, 'representation':'object', 'remove_outliers':False, 'verbose':True} _, _ = _qc_plot(wells_per_gene_settings) grna_well_data_path = os.path.join(res_folder, 'well_grna.csv') prc_gene_count_df.to_csv(grna_well_data_path, index=False) print(f"Saved well per grna data to {grna_well_data_path}") grna_per_well_settings = {'src':grna_well_data_path, 'graph_name':'gene_per_well', 'data_column':['gene_count'], 'grouping_column':'plateID', 'graph_type':qc_graph_type, 'theme':'bright', 'save':True, 'y_lim':[None,None], 'log_y':False, 'log_x':False, 'representation':'well', 'remove_outliers':False, 'verbose':False} _, _ = _qc_plot(grna_per_well_settings) except Exception as e: print(e) if str(settings.get('inference', 'parametric')).lower() == 'auto': resolved_mode, reason = resolve_auto_inference(merged_df, settings) settings['analysis_mode'] = resolved_mode print(f"inference='auto': {reason}") elif settings.get('analysis_mode') == 'regression': for _one in resolve_levels(settings.get('regression_type'), settings.get('level', 'both')): warning = _identifiability_warning(merged_df, settings, level=_one) if warning: print(f" level={_one!r}:") print(warning) if settings.get('analysis_mode') == 'guide_permutation': _chosen = settings.get('regression_type') if _chosen: print(f"inference='nonparametric': this is a permutation test, so " f"it fits no model and regression_type={_chosen!r} is not " f"read. Choosing a different regression_type with this " f"inference gives the same numbers; set " f"inference='parametric' to fit {_chosen!r} itself.") _stage(settings, "permuting the guides") output = _run_guide_permutation_analysis( merged_df, dependent_variable, res_folder, settings) _stage(settings, "the permutation has returned") if settings.get('verbose'): print( f"Guide permutation analysis tested " f"{len(output['primary'])} guides in the primary " f">={output['primary_min_wells']}-well family and called " f"{len(output['significant'])} at " f"{settings['multiple_testing_method']} " f"alpha={settings['fdr_alpha']}." ) try: from .regression_summary import write_run_summary except ImportError: pass else: try: write_run_summary( res_folder, model=None, settings=settings, coef_df=output.get('primary'), regression_type=settings.get('regression_type'), fit_designs={}) except Exception as error: # noqa: BLE001 - never lose a run print(f"Could not write the run summary: " f"{type(error).__name__}: {error}") output.setdefault('res_folder', res_folder) output.setdefault('settings', dict(settings)) output.setdefault('regression_type', settings.get('regression_type')) return output if not _show_plates(merged_df, orig_dv, res_folder): _ = plot_plates(merged_df, variable=orig_dv, grouping='mean', min_max='allq', cmap='viridis', min_count=None, dst=res_folder) _stage(settings, "fitting the model") fits = regression_levels( merged_df, csv_path, dependent_variable=dependent_variable, regression_type=settings['regression_type'], regression_backend=settings.get('regression_backend', DEFAULT_REGRESSION_BACKEND), level=settings.get('level', 'both'), alpha=settings['alpha'], random_row_column_effects=settings['random_row_column_effects'], model_plate_position=settings.get('model_plate_position', True), model_data_layout=settings.get('model_data_layout', 'long'), nc=settings['negative_control_id'], pc=settings['positive_control_id'], controls=settings['nontargeting_control_grnas'], dst=res_folder, verbose=bool(settings.get('verbose')), transform=str(settings.get('transform') or ''), cov_type=settings['cov_type'], l1_ratio=settings['l1_ratio'], quantile=settings['quantile'], hinge_threshold=settings['hinge_threshold'], hinge_n_boot=settings['hinge_n_boot'], huber_t=settings['huber_t'], spline_knots=settings.get('spline_knots', 4), spline_degree=settings.get('spline_degree', 3), group_lasso_lambda=settings.get('group_lasso_lambda', 'auto'), rra_alpha=settings.get('rra_alpha', 0.25), rra_permutations=settings.get('rra_permutations', 10000), qc=bool(settings.get('regression_qc', True)), legacy_volcano=bool(settings.get('legacy_volcano', False)), intercept=str(settings.get('intercept') or 'fitted'), intercept_value=float(settings.get('intercept_value') or 0.0), ) regression_type = next(iter(fits.values()))[2] fit_designs = {one: dict(one_coef.attrs.get('fit_design', {})) for one, (_model, one_coef, _type) in fits.items()} settings['_regression_diagnostics'] = _write_regression_diagnostics( res_folder, merged_df, fits, settings) level_tables = { one: _annotate_level_coefficients(one_coef, n_grna, n_gene) for one, (_model, one_coef, _type) in fits.items() } for table in level_tables.values(): table.attrs.pop('fit_design', None) if regression_type == 'mixed' and 'gene' in level_tables: whole = level_tables.pop('gene') blups = whole['term_type'] == TERM_BLUP level_tables['gene'] = whole.loc[~blups].copy() guide_table = whole.loc[blups].copy() guide_table['level'] = 'grna' guide_table['q_value'] = np.nan guide_table['multiple_testing_method'] = 'none' level_tables['grna'] = guide_table print(f"Mixed fit: {len(level_tables['gene'])} gene rows corrected as " f"one family, {len(guide_table)} guide BLUPs written without a " f"q value. Choose a fixed-effects model with level='grna' for a " f"guide-level hit list.") corrected = {} hits_by_level = {} thresholds_by_level = {} for one, table in level_tables.items(): if regression_type == 'mixed' and one == 'grna': corrected[one] = table hits_by_level[one] = table.iloc[0:0] thresholds_by_level[one] = 0 continue table, level_hits, level_threshold, _rule = _call_level_hits( table, one, settings, regression_type, merged_df, dependent_variable, bootstrap=bootstrap_selection_frequencies) corrected[one] = table hits_by_level[one] = level_hits thresholds_by_level[one] = level_threshold primary = 'grna' if 'grna' in fits else next(iter(fits)) model = fits[primary][0] reg_threshold = thresholds_by_level.get(primary, 0) grna_coef_df = corrected.get('grna') gene_coef_df = corrected.get('gene') if grna_coef_df is not None: grna_coef_df = grna_coef_df.dropna(subset=['n_grna']) if gene_coef_df is not None: gene_coef_df = gene_coef_df.dropna(subset=['n_gene']) template = corrected[primary].iloc[0:0] if grna_coef_df is None: print("level='gene': no guide fit was run, so results_grna.csv is " "written empty. Set level='both' or level='grna' for one.") grna_coef_df = template if gene_coef_df is None: print("level='grna': no gene fit was run, so results_gene.csv is " "written empty. Set level='both' or level='gene' for one.") gene_coef_df = template def _stack(frames): """Concatenate nonempty frames, or return the empty result template.""" kept = [frame for frame in frames if len(frame)] return pd.concat(kept, ignore_index=True) if kept else template coef_df = _stack(corrected.values()) significant = _stack(hits_by_level.values()) if _annotation_source(settings): from .annotation import annotate_with, supplementary source = _annotation_source(settings) cache = _annotation_cache(settings) annotated, notes = {}, [] for name, frame in (('results', coef_df), ('gene', gene_coef_df), ('grna', grna_coef_df), ('significant', significant)): before = len(frame) annotated[name], note = annotate_with( frame, source, cache_dir=cache, quiet=(name != 'results')) if note and name == 'results': notes.append(note) if len(annotated[name]) != before: raise ValueError( f"the {source} annotation changed {name} from {before} " f"to {len(annotated[name])} row(s).") for note in notes: print(f"Annotation: {note}") coef_df = annotated['results'] gene_coef_df = annotated['gene'] grna_coef_df = annotated['grna'] significant = annotated['significant'] supplementary( coef_df['feature'] if 'feature' in coef_df.columns else None, path=os.path.join(res_folder, 'supplementary_topology.csv')) coef_df.to_csv(results_path, index=False) gene_coef_df.to_csv(results_path_gene, index=False) grna_coef_df.to_csv(results_path_grna, index=False) if regression_type in ['ols', 'beta']: if settings['verbose']: print(model.summary()) save_summary_to_file( model, file_path=os.path.join(res_folder, SUMMARY_FILENAME)) try: from .regression_summary import write_run_summary except ImportError: pass else: try: _stage(settings, "the fit has returned") write_run_summary(res_folder, model=model, settings=settings, coef_df=coef_df, regression_type=regression_type, fit_designs=fit_designs) except Exception as error: # noqa: BLE001 - never lose a run print(f"Could not write the run summary: " f"{type(error).__name__}: {error}") significant.to_csv(hits_path, index=False) threshold = settings['min_observations_per_hit'] significant_grna_filtered = significant[significant['n_grna'] > threshold] significant_gene_filtered = significant[significant['n_gene'] > threshold] significant_filtered = pd.concat([significant_grna_filtered, significant_gene_filtered]) filtered_hit_path = os.path.join(os.path.dirname(hits_path), 'results_significant_filtered.csv') significant_filtered.to_csv(filtered_hit_path, index=False) if isinstance(settings['metadata_files'], str): settings['metadata_files'] = [settings['metadata_files']] results_metadata_df = tabular.read_table(results_path, report=None) gene_merged_df = tabular.read_table(results_path_gene, report=None) grna_merged_df = tabular.read_table(results_path_grna, report=None) for metadata_file in settings['metadata_files']: file = os.path.basename(metadata_file) filename, _ = os.path.splitext(file) try: if not os.path.isfile(metadata_file) \ or os.path.getsize(metadata_file) == 0: print(f"Skipping empty or missing metadata file: " f"{metadata_file}") continue except OSError: continue try: _ = merge_regression_res_with_metadata(hits_path, metadata_file, name=filename) results_metadata_df = merge_regression_res_with_metadata(results_path, metadata_file, name=filename) gene_merged_df = merge_regression_res_with_metadata(results_path_gene, metadata_file, name=filename) grna_merged_df = merge_regression_res_with_metadata(results_path_grna, metadata_file, name=filename) except Exception as metadata_error: print(f"Could not merge metadata from {metadata_file}: " f"{metadata_error}") continue draw_legacy_volcano = bool(settings.get('legacy_volcano', False)) if not draw_legacy_volcano: print("Legacy volcano: off (the interactive volcano and the house-" "style figure are drawn instead). Set legacy_volcano=True to " "draw the original matplotlib one as well.") if _toxoplasma_is_on(settings): data_path = results_metadata_df data_path_gene = gene_merged_df data_path_grna = grna_merged_df base_dir = os.path.dirname(os.path.abspath(__file__)) metadata_path = os.path.join(base_dir, 'resources', 'data', 'lopit.csv') gene_list = custom_volcano_plot( gene_merged_df, metadata_path, metadata_column='tagm_location', point_size=600, figsize=20, threshold=reg_threshold, save_path=volcano_path, x_lim=settings.get('x_lim'), y_lims=settings.get('y_lims'), draw=draw_legacy_volcano, ) if not draw_legacy_volcano: pass elif os.path.exists(volcano_path): print(f"Saved volcano plot to {volcano_path}") else: print(f"WARNING: the legacy volcano was requested but no file was " f"written to {volcano_path}") display(gene_list) if gene_list is not None else None phenotype_plot = os.path.join(res_folder, 'phenotype_plot.pdf') transcription_heatmap = os.path.join(res_folder, 'transcription_heatmap.pdf') metadata_files = list(settings.get('metadata_files') or []) have_curated_tables = len(metadata_files) >= 2 if not have_curated_tables: print(f"Skipping the phenotype and transcription reports: they " f"need two curated metadata tables (GT1 phenotypes and " f"ME49 expression) and {len(metadata_files)} were given. " f"The volcano and every results table are unaffected.") data_GT1 = (tabular.read_table(metadata_files[1], low_memory=False, canonicalise=False, report=None) if have_curated_tables else None) data_ME49 = (tabular.read_table(metadata_files[0], low_memory=False, canonicalise=False, report=None) if have_curated_tables else None) columns = ['sense - Tachyzoites', 'sense - Tissue cysts', 'sense - EES1', 'sense - EES2', 'sense - EES3', 'sense - EES4', 'sense - EES5'] if gene_list and have_curated_tables: print('Plotting gene phenotypes and heatmaps') print(gene_list) plot_gene_phenotypes(data=data_GT1, gene_list=gene_list, save_path=phenotype_plot) plot_gene_heatmaps( data=data_ME49, gene_list=gene_list, columns=columns, x_column='Gene ID', normalize=True, save_path=transcription_heatmap, ) elif not gene_list: print("No gene_list produced; skipping phenotype and heatmap plots.") if not _toxoplasma_is_on(settings) and draw_legacy_volcano: try: from .plot import volcano_plot as _plain_volcano _source = results_path_gene _plain_volcano( _source, fold_change_col='coefficient', p_value_col='p_value', name_col='feature', x_transform='none', y_transform='-log10', fold_change_threshold=reg_threshold, p_value_threshold=float(settings.get('fdr_alpha', 0.05) or 0.05), point_size=20.0, figsize=(10.0, 8.0), title=f"{settings.get('regression_type', 'ols')} - gene", save_path=volcano_path, show=False) except Exception as _volcano_error: print(f"Could not draw the volcano plot: " f"{type(_volcano_error).__name__}: {_volcano_error}") if os.path.exists(volcano_path): print(f"Saved volcano plot to {volcano_path}") print('Significant Genes') grnas = significant['grna'].unique().tolist() genes = significant['gene'].unique().tolist() print(f"Found p<0.05 coedfficients for {len(grnas)} gRNAs and {len(genes)} genes") display(significant) _warn_if_penalised_no_hits(settings, coef_df) try: from .guide_concordance import concordance_report controls = {} for _key, _role in (('positive_control_id', 'positive'), ('negative_control_id', 'negative')): _value = settings.get(_key) if _value not in (None, ''): controls[str(_value)] = _role print() print(concordance_report( coef_df, alpha=float(settings.get('fdr_alpha', 0.05) or 0.05), controls=controls)) except Exception as concordance_error: print(f"Could not summarise guide support: {concordance_error}") output = {'results':coef_df, 'significant':significant, 'model': model, 'model_data': merged_df, 'fit_designs': fit_designs, 'regression_type': regression_type, 'res_folder': res_folder, 'settings': dict(settings)} manifest = getattr(coef_df, "attrs", {}).get("qc_manifest") if manifest: output['qc'] = manifest worst = manifest.get('verdict') if worst is not None: output['qc_verdict'] = worst output['qc_verdict_level'] = manifest.get('verdict_level', 'unknown') return output #: The fixed head of a ``prcfo`` key, in order. The object id is always the #: LAST token and anything between the two is the timepoint, which is how a #: five-token and a six-token key are told apart without guessing. _PRCFO_HEAD = schema.FIELD_KEY_COLUMNS def _assign_prcfo_parts(df, object_column='objectID'): """Split ``prcfo`` into its named components and assign them onto ``df``. ``prcfo`` is written by :func:`spacr.utils._map_wells_png` and rebuilt by :func:`spacr.utils._split_data`. It has **five** tokens on a plain screen (``plate_row_column_field_object``) and **six** on a timelapse (``plate_row_column_field_TIME_object``). Three places in this module used to spell that as .. code-block:: python df[['plateID', 'rowID', 'columnID', 'fieldID', 'objectID']] = \\ df['prcfo'].str.split('_', expand=True) which is not a mis-assignment on a timelapse — it is a hard stop. Six split columns against five keys makes pandas raise ``ValueError: Columns must be same length as key``, so :func:`ml_analysis` threw away a completed model at its very last statement (measured on a real 2-well x 2-field x 3-frame x 3-object database: 36 rows in, fit and permutation importance done, then ``ValueError`` at ``ml.py:2517``). The five names would *also* have been wrong had it not raised — the fifth token of a timelapse key is the timepoint, so ``objectID`` would have held ``'t1'`` and the object id would have been dropped entirely. Splitting the head from the left and the object from the right recovers both forms, and the timepoint is kept rather than discarded: it is written under whichever spelling ``df`` already uses (``timeID`` canonical, ``time_id`` legacy — resolved through :func:`spacr.utils._time_column`), defaulting to ``timeID``. This doubles as repair-on-read for a scores CSV whose ``objectID`` was filled in by a positional guess over a timelapse crop name — the same guess :func:`spacr.ml.interperate_vision_model` already refuses to trust — because the components are recomputed from ``prcfo`` and overwrite what is there. :param df: Frame carrying a ``prcfo`` column. :param object_column: Name to give the object id. ``'objectID'`` for the read/score paths, ``'object'`` in :func:`ml_analysis`, which is what each of them already wrote. :returns: ``df``, with the component columns assigned. :raises TimelapseKeyMismatch: when the frame mixes five- and six-token keys — two runs that disagreed about ``timelapse`` were concatenated, and there is no single answer to what the fifth token means. :raises ValueError: when a key has neither five nor six tokens. """ from .io import TimelapseKeyMismatch from .utils import _time_column values = df['prcfo'] tokens = values.astype(str).str.split(schema.KEY_SEPARATOR) widths = tokens.map(len) seen = set(widths.unique().tolist()) unexpected = sorted(seen - {5, 6}) if unexpected: example = values[widths.isin(unexpected)].iloc[0] raise ValueError( f"prcfo must be plate_row_column_field_object (5 tokens) or " f"plate_row_column_field_time_object (6, timelapse); found " f"{unexpected} token(s), e.g. {example!r}." ) if seen == {5, 6}: raise TimelapseKeyMismatch( f"prcfo mixes {int((widths == 5).sum())} key(s) without a " f"timepoint and {int((widths == 6).sum())} with one, so the fifth " f"token is an object id in some rows and a timepoint in others. " f"Two runs that disagreed about 'timelapse' have been combined; " f"re-run the non-timelapse half rather than splitting this." ) parsed = [schema.parse_prcfo(value) for value in values.astype(str)] for name in _PRCFO_HEAD: df[name] = [getattr(obj, name) for obj in parsed] df[object_column] = [obj.objectID for obj in parsed] if seen == {6}: df[_time_column(df.columns) or schema.TIME_KEY] = [ obj.timeID for obj in parsed] return df
[docs] def process_reads(csv_path, fraction_threshold, plate, filter_column=None, filter_value=None, record=None, exclude_grnas=None): """Load a per-gRNA read-count CSV and return per-well normalised fractions. Splits derived ``plate_row`` or ``prcfo`` identifiers, computes each gRNA's fraction of the well total, applies an optional fraction-cutoff filter and returns a compact ``(prc, grna, fraction)`` frame (with ``gene`` derived from the gRNA when possible). :param csv_path: Path to the counts CSV, or an already-loaded DataFrame. :param fraction_threshold: Drop rows below this fraction; must be in ``[0, 1]`` or ``None``. :param plate: Plate identifier used when no ``plateID`` column is present. :param filter_column: Column (or list of columns) to filter rows on. :param filter_value: Values (or list of values) to drop from ``filter_column``. :param record: Optional mutable mapping that records exclusions for the persisted regression summary. :param exclude_grnas: Guide or gene identifiers to remove from the raw count table. Gene identifiers remove all associated guides. This is applied before well totals and fractions are calculated, so retained guides are normalised against the retained read count. :returns: DataFrame with columns ``prc``, ``grna``, ``fraction``. :raises ValueError: on missing required columns, invalid ``fraction_threshold``, or when the threshold removes all rows. """ from .utils import correct_metadata if isinstance(csv_path, pd.DataFrame): csv_df = csv_path else: csv_df = tabular.read_table(csv_path) csv_df = correct_metadata(csv_df) if 'grna_name' in csv_df.columns: csv_df = csv_df.rename(columns={'grna_name': 'grna'}) if exclude_grnas and 'grna' in csv_df.columns: from .read_background import resolve_exclusions, unmatched_exclusions requested = ([exclude_grnas] if isinstance(exclude_grnas, str) else list(exclude_grnas)) guide_names = csv_df['grna'].astype(str) gene_names = (csv_df['gene'].astype(str) if 'gene' in csv_df.columns else None) resolved = resolve_exclusions(requested, guide_names, gene_names) unmatched = unmatched_exclusions(requested, guide_names, gene_names) drop_mask = guide_names.isin(resolved) rows_before = len(csv_df) rows_removed = int(drop_mask.sum()) if rows_removed: csv_df = csv_df.loc[~drop_mask].copy() print( f"Excluded {rows_removed} of {rows_before} raw count rows " f"spanning {len(resolved)} guide(s) named by exclude_grnas " f"before well totals and fractions were calculated: " f"{', '.join(sorted(resolved)[:5])}" f"{' ...' if len(resolved) > 5 else ''}." ) if unmatched: print( f"exclude_grnas named {len(unmatched)} value(s) that match " f"no guide or gene in the raw count table: " f"{', '.join(map(str, unmatched[:5]))}" f"{' ...' if len(unmatched) > 5 else ''}." ) if record is not None: record["exclude_grnas"] = ( record.get("exclude_grnas", 0) + rows_removed) record["exclude_grnas_of"] = ( record.get("exclude_grnas_of", 0) + rows_before) prior_guides = record.get("exclude_grnas_guides", ()) record["exclude_grnas_guides"] = sorted( set(map(str, prior_guides)) | set(map(str, resolved))) prior_unmatched = record.get("exclude_grnas_unmatched", ()) record["exclude_grnas_unmatched"] = sorted( set(map(str, prior_unmatched)) | set(map(str, unmatched))) if csv_df.empty: raise ValueError( f"exclude_grnas removed all {rows_before} raw count rows. " "Remove or narrow the exclusion before running regression." ) if 'plate_row' in csv_df.columns: pieces = csv_df['plate_row'].astype(str).str.rsplit( schema.KEY_SEPARATOR, n=1) malformed = pieces.map(len) < 2 if malformed.any(): example = csv_df.loc[malformed, 'plate_row'].iloc[0] raise ValueError( f"'plate_row' must be '<plate>{schema.KEY_SEPARATOR}<row>', " f"but {int(malformed.sum())} of {len(csv_df)} value(s) hold no " f"{schema.KEY_SEPARATOR!r}, e.g. {example!r}. Supply separate " f"'plateID' and 'rowID' columns instead, or repair the count " f"table — guessing which half is the plate would key the whole " f"screen on the wrong well.") csv_df['plateID'] = pieces.str[0] csv_df['rowID'] = pieces.str[-1] if not 'plateID' in csv_df.columns: if not plate is None: csv_df['plateID'] = plate else: csv_df['plateID'] = 'plate1' if 'prcfo' in csv_df.columns: csv_df = _assign_prcfo_parts(csv_df, object_column='objectID') csv_df['prc'] = _compose_prc_column(csv_df) if isinstance(filter_column, str): filter_column = [filter_column] if isinstance(filter_value, str): filter_value = [filter_value] if isinstance(filter_column, list): for filter_col in filter_column: for value in filter_value: csv_df = csv_df.loc[csv_df[filter_col] != value].copy() if not all(col in csv_df.columns for col in ['rowID','columnID','grna','count']): raise ValueError("The CSV file must contain 'grna', 'count', 'rowID', and 'columnID' columns.") csv_df['prc'] = _compose_prc_column(csv_df) grouped_df = csv_df.groupby('prc')['count'].sum().reset_index() grouped_df = grouped_df.rename(columns={'count': 'total_counts'}) merged_df = pd.merge(csv_df, grouped_df, on='prc', validate='many_to_one') merged_df['fraction'] = merged_df['count'] / merged_df['total_counts'] if fraction_threshold is not None: if not 0 <= fraction_threshold <= 1: raise ValueError( f"fraction_threshold={fraction_threshold} is outside the valid range [0, 1]. " f"The 'fraction' column is a relative abundance bounded between 0 and 1." ) observations_before = len(merged_df) frac_min = merged_df['fraction'].min() frac_max = merged_df['fraction'].max() frac_median = merged_df['fraction'].median() merged_df = merged_df[merged_df['fraction'] >= fraction_threshold] observations_after = len(merged_df) removed = observations_before - observations_after if record is not None: record["fraction_threshold"] = ( record.get("fraction_threshold", 0) + int(removed)) record["fraction_threshold_of"] = ( record.get("fraction_threshold_of", 0) + int(observations_before)) pct_retained = 100 * observations_after / observations_before if observations_before else 0 print( f"Removed {removed} of {observations_before} observations " f"below fraction threshold {fraction_threshold} " f"({pct_retained:.1f}% retained). " f"Fraction range in input: [{frac_min:.4g}, {frac_max:.4g}], median {frac_median:.4g}." ) if observations_after == 0: raise ValueError( f"All {observations_before} rows were removed by fraction_threshold={fraction_threshold}. " f"Observed fraction range was [{frac_min:.4g}, {frac_max:.4g}], median {frac_median:.4g}. " f"Choose a threshold below the median, or pass None to auto-compute." ) merged_df = merged_df[['prc', 'grna', 'fraction']] tokens = merged_df['grna'].astype(str).str.split(schema.KEY_SEPARATOR) widths = sorted(set(tokens.map(len).tolist())) if widths == [3]: merged_df['gene'] = tokens.str[1] merged_df['grna'] = (tokens.str[1] + schema.KEY_SEPARATOR + tokens.str[2]) else: example = merged_df['grna'].iloc[0] if len(merged_df) else None print(f"Not splitting 'grna' into org/gene/grna: that split is " f"positional and needs every name to be " f"'<org>{schema.KEY_SEPARATOR}<gene>{schema.KEY_SEPARATOR}" f"<guide>' (3 components), but this table holds names with " f"{widths} component(s), e.g. {example!r}. No 'gene' column " f"is produced; a step that needs one will name it.") return merged_df
#: The squeeze applied before a logit, and the reason it exists. #: #: A classification score is a PROPORTION, and a screen produces exact 0 and #: exact 1 -- neither of which has a logit. Smithson and Verkuilen's transform #: pulls the whole scale off the endpoints by (n-1)/n plus a half, which is #: the standard treatment and is reported in the run summary rather than #: applied quietly: a transform that silently moved a user's 0 to 0.001 #: changed their data. BETA_SQUEEZE_NOTE = ( "beta: the response was mapped to the logit scale. A proportion of " "exactly 0 or 1 has no logit, so the scale was squeezed off its " "endpoints by the Smithson-Verkuilen rule ((y*(n-1)+0.5)/n) first")
[docs] def beta_logit(values): """A proportion on the logit scale, with the endpoints squeezed in. ``transform='beta'`` is intended for proportional responses such as classification scores and their well aggregates, where a logarithm is not appropriate. This is distinct from ``regression_type='beta'``, which selects a beta GLM. One transforms the response; the other selects the model family. :param values: proportions in ``[0, 1]``, array-like; converted to a float array. Non-finite entries pass through unchanged. When any finite value is at or beyond 0 or 1 the finite values are squeezed with ``(y * (n - 1) + 0.5) / n`` before the logit. """ array = np.asarray(values, dtype=float) finite = np.isfinite(array) n = int(finite.sum()) if n < 1: return array squeezed = array.copy() inside = array[finite] if inside.min() <= 0.0 or inside.max() >= 1.0: squeezed[finite] = (inside * (n - 1) + 0.5) / n squeezed[finite] = np.clip(squeezed[finite], 1e-9, 1.0 - 1e-9) out = np.array(array, dtype=float, copy=True) out[finite] = np.log(squeezed[finite] / (1.0 - squeezed[finite])) return out
[docs] def apply_transformation(X, transform): """Return an sklearn ``FunctionTransformer`` for the named transform. :param X: Ignored (kept for compatibility with sklearn pipeline flow). :param transform: One of ``'log'``, ``'sqrt'``, ``'square'``, ``'beta'``. Any other value returns ``None``. :returns: A ``FunctionTransformer`` or ``None``. """ if transform == 'log': transformer = FunctionTransformer(np.log1p, validate=True) elif transform == 'sqrt': transformer = FunctionTransformer(np.sqrt, validate=True) elif transform == 'square': transformer = FunctionTransformer(np.square, validate=True) elif transform == 'beta': transformer = FunctionTransformer(beta_logit, validate=True) else: transformer = None return transformer
[docs] def check_normality(data, variable_name, verbose=False): """Check if the data is normally distributed using the Shapiro-Wilk test. :param data: numeric values, array-like; non-finite values are dropped and fewer than 3 remaining values returns ``False`` without testing. :param variable_name: name printed in the verbose messages only. :param verbose: print the test statistic, P value and verdict. :returns: ``True`` when the Shapiro-Wilk P value exceeds 0.05. """ values = np.asarray(data, dtype=float) values = values[np.isfinite(values)] if values.size < 3: if verbose: print(f"Shapiro-Wilk Test for {variable_name}: at least 3 finite " f"values are required; received {values.size}.") return False stat, p_value = shapiro(values) if verbose: print(f"Shapiro-Wilk Test for {variable_name}:\nStatistic: {stat}, P-value: {p_value}") if p_value > 0.05: if verbose: print(f"Normal distribution: The data for {variable_name} is normally distributed.") return True else: if verbose: print(f"Normal distribution: The data for {variable_name} is not normally distributed.") return False
[docs] def clean_controls(df,values, column): """Drop rows whose ``column`` holds one of the listed ``values``. :param df: Source DataFrame. :param values: List of values to remove. Anything that is not a list (a bare value included) is a no-op. :param column: Column, or list of columns, to check. ``None`` is a no-op. :returns: Filtered DataFrame (unchanged if ``column`` is missing or ``values`` is not a list). """ if column is None: return df columns = list(column) if isinstance(column, (list, tuple, set)) else [column] if isinstance(values, list): for col in columns: if col in df.columns: for value in values: df = df[~df[col].isin([value])] print(f'Removed data from {value}') return df
[docs] def process_scores(df, dependent_variable, plate, min_cells_per_well=25, agg_type='mean', transform=None, regression_type='ols', invert_dependent_variable=False): """Aggregate per-object model scores to per-well summaries, ready for regression. Ensures ``plateID/rowID/columnID/prc`` columns exist, applies an optional inversion of the raw response, aggregates by well according to ``agg_type`` (or with ``sum`` for the count models ``'poisson'`` and ``'horseshoe'``), enforces ``min_cells_per_well`` and optionally transforms the aggregated response. :param df: Per-object score DataFrame. :param dependent_variable: Column being aggregated. :param plate: Plate identifier to stamp when the frame is single-plate; ignored (with warning) when multiple plates exist. :param min_cells_per_well: Wells with fewer objects are dropped. Default ``25``. :param agg_type: ``'mean'``, ``'median'``, ``'quantile'`` or None. :param transform: Optional post-aggregation transform name (see :func:`apply_transformation`). :param regression_type: If ``'poisson'`` or ``'horseshoe'``, aggregation uses ``sum`` - both model a per-well count, not a per-well average. :param invert_dependent_variable: ``False``/``0`` = no inversion; ``True``/``1`` = ``1 - x``; ``-1`` = ``1 / x``. :returns: ``(dependent_df, dependent_variable)`` — the per-well DataFrame and the (possibly transformed) response column name. :raises ValueError: on missing identifiers, unsupported ``agg_type`` or unrecognised ``invert_dependent_variable``. """ from .utils import correct_metadata df = df.reset_index(drop=True) if 'prcfo' in df.columns: df = df.loc[:, ~df.columns.duplicated()].copy() if not all(col in df.columns for col in ['plateID', 'rowID', 'columnID']): df = _assign_prcfo_parts(df, object_column='objectID') df['prc'] = _compose_prc_column(df) else: df = correct_metadata(df) df = df.loc[:, ~df.columns.duplicated()].copy() n_plates_in_df = df['plateID'].nunique(dropna=True) if 'plateID' in df.columns else 0 if plate is not None: if n_plates_in_df > 1: print(f"Warning: process_scores received plate={plate!r} but the input " f"DataFrame already contains {n_plates_in_df} distinct plateIDs. " f"Ignoring the 'plate' argument and using the per-row plateID " f"column to avoid collapsing plates.") else: df['plateID'] = plate if 'plateID' not in df.columns or df['plateID'].isna().all(): raise ValueError( "process_scores: DataFrame has no usable 'plateID' column " "and no 'plate' argument was provided." ) if all(col in df.columns for col in ['plateID', 'rowID', 'columnID']): df['prc'] = _compose_prc_column(df) else: raise ValueError("The DataFrame must contain 'plateID', 'rowID', and 'columnID' columns.") df = df[['prc', dependent_variable]] df = df[['prc', dependent_variable]].copy() if invert_dependent_variable in (True, 1): df[dependent_variable] = 1.0 - df[dependent_variable] print(f"Inverted '{dependent_variable}' as 1 - x on raw values.") elif invert_dependent_variable == -1: raw = df[dependent_variable] n_zero = int((raw == 0).sum()) if n_zero > 0: print(f"Warning: '{dependent_variable}' contains {n_zero} zero " f"values; 1/x is undefined for those rows. They will be set " f"to NaN and dropped from this analysis.") df[dependent_variable] = 1.0 / raw.where(raw != 0) df = df.dropna(subset=[dependent_variable]) print(f"Inverted '{dependent_variable}' as 1/x on raw values.") elif invert_dependent_variable in (False, 0): pass else: raise ValueError( f"invert_dependent_variable must be one of False, True, 1, -1; " f"got {invert_dependent_variable!r}." ) grouped = df.groupby('prc')[dependent_variable] count_models = ('poisson', 'horseshoe') if regression_type not in count_models: print(f'Using agg_type: {agg_type}') if agg_type == 'median': dependent_df = grouped.median().reset_index() elif agg_type == 'mean': dependent_df = grouped.mean().reset_index() elif agg_type == 'quantile': dependent_df = grouped.quantile(0.75).reset_index() elif agg_type is None: dependent_df = df.reset_index() if 'prcfo' in dependent_df.columns: dependent_df = dependent_df.drop(columns=['prcfo']) else: raise ValueError(f"Unsupported aggregation type {agg_type}") if regression_type in count_models: agg_type = 'count' print(f'Using agg_type: {agg_type} for {regression_type} regression') dependent_df = grouped.sum().reset_index() summed = pd.to_numeric(dependent_df.get(dependent_variable), errors='coerce') if summed is not None and len(summed): finite = summed[np.isfinite(summed)] if len(finite) and not np.all( np.isclose(finite, np.rint(finite), rtol=0, atol=1e-8)): example = float(finite.iloc[0]) raise ValueError( f"regression_type={regression_type!r} models the well's " f"positive COUNT -- the number of cells called positive " f"-- and gets it by summing {dependent_variable!r} per " f"well. That column holds continuous scores, so the sum " f"is {example:.4g} rather than a whole number of cells. " f"Either fit a continuous model ('ols', 'mixed', or " f"'beta', which is built for a proportion), or give " f"dependent_variable a per-cell 0/1 label so its " f"per-well sum is a real count.") cell_count = grouped.size().reset_index(name='cell_count') if agg_type is None: dependent_df = pd.merge(dependent_df, cell_count, on='prc', validate='many_to_one') else: dependent_df['cell_count'] = cell_count['cell_count'] print("1 test") display(dependent_df) dependent_df = dependent_df[dependent_df['cell_count'] >= min_cells_per_well] print("2 test") display(dependent_df) is_normal = check_normality(dependent_df[dependent_variable], dependent_variable) if transform is not None and regression_type in count_models: print(f"Ignoring transform={transform!r}: {regression_type} models a " f"per-well count, and a transformed count is not a count.") transform = None if transform == 'beta': column = pd.to_numeric(dependent_df[dependent_variable], errors='coerce') inside = column[np.isfinite(column)] if len(inside) and (inside.min() < 0.0 or inside.max() > 1.0): raise ValueError( f"transform='beta' puts the response on the logit scale, " f"which is only defined for a proportion, but " f"{dependent_variable!r} runs from {inside.min():.4g} to " f"{inside.max():.4g}. Use a score or a fraction here, or " f"pick transform='log' for a response in measured units.") print(BETA_SQUEEZE_NOTE) if transform is not None: transformer = apply_transformation(dependent_df[dependent_variable], transform=transform) transformed_var = f'{transform}_{dependent_variable}' dependent_df[transformed_var] = transformer.fit_transform(dependent_df[[dependent_variable]]) dependent_variable = transformed_var is_normal = check_normality(dependent_df[transformed_var], transformed_var) if not is_normal: print(f'{dependent_variable} is not normally distributed') else: print(f'{dependent_variable} is normally distributed') return dependent_df, dependent_variable
@single_threaded_openmp('classical ML training') @_flowview_pipeline("ml")
[docs] def generate_ml_scores(settings): """Train a classical ML classifier (XGBoost / logistic / RF) on per-object features and score every well of a screen. Reads the measurement store selected by ``measurement_backend`` from :func:`spacr.measure.measure_crop`, merges cell/nucleus/pathogen/ cytoplasm feature tables, uses the wells marked as ``positive_control`` / ``negative_control`` (or an annotation column) as training labels, delegates fitting to :func:`ml_analysis`, and writes per-object predictions, permutation and feature-importance tables plus a plate heatmap into ``results/`` under the first source folder. :param settings: Settings dict, canonicalized via :func:`spacr.settings.set_default_analyze_screen`. Key entries: - ``src`` (str or list) — folder(s) containing the measurements. - ``measurement_backend`` / ``measurement_backend_target`` — select the SQLite, DuckDB, Parquet or PostgreSQL measurement store. - ``channel_of_interest`` — 0-based channel for the recruitment ratio feature; also drives table selection. - ``model_type_ml`` — ``'xgboost'``, ``'logistic_regression'``, ``'random_forest'``. - ``positive_control`` / ``negative_control`` — well IDs (e.g. ``'c2'`` / ``'c1'``) used as training labels. - ``annotation_column`` — override controls with a PNG-level annotation column. - ``location_column`` — ``'columnID'`` or ``'rowID'``. - ``heatmap_feature`` — feature plotted on the plate heatmap. - ``exclude``, ``n_repeats``, ``top_features``, ``test_size``, ``reg_alpha``, ``reg_lambda``, ``learning_rate``, ``n_estimators``, ``n_jobs``. - ``remove_low_variance_features``, ``remove_highly_correlated_features``, ``prune_features``, ``cross_validation``, ``verbose``. :returns: The two-element list ``[output, plate_heatmap]``, where ``output`` is the 10-element result list of :func:`ml_analysis` and ``plate_heatmap`` is the plate-heatmap ``matplotlib`` figure. The CSVs and figures are written to ``results/`` as a side effect; their paths are not returned. :raises ValueError: if ``annotation_column`` is set but the ``png_list`` table lacks ``prcfo`` / that column, its object IDs do not join to the measurements, it contains fewer than two observed classes, or if ``heatmap_feature`` is not among the trained features. Example: .. code-block:: python from spacr.ml import generate_ml_scores settings = { 'src': '/data/plate01', 'channel_of_interest': 3, 'positive_control_id': 'c2', 'negative_control_id': 'c1', 'model_type_ml': 'xgboost', 'heatmap_feature': 'recruitment', } generate_ml_scores(settings) See Also: :func:`ml_analysis` — the underlying fit/evaluate routine. :func:`perform_regression` — mixed-effects regression on per-well ML scores. """ from .io import _read_and_merge_data, _read_db from .plot import plot_plates from .utils import (get_ml_results_paths, calculate_shortest_distance, save_settings, _measurement_store_for) from .settings import set_default_analyze_screen from .predictions import (ML_CLASS_COLUMN, merge_ml_predictions, migrate_prediction_columns) settings = set_default_analyze_screen(settings) save_settings(settings, name='generate_ml_scores', show=True) _flowview_advance("tables") srcs = settings['src'] if isinstance(srcs, str): srcs = [srcs] df = pd.DataFrame() measurement_stores = [] for idx, src in enumerate(srcs): if idx == 0: src1 = src sqlite_path = os.path.join(src, 'measurements', 'measurements.db') db_loc = [_measurement_store_for(sqlite_path, settings) or sqlite_path] if (str(settings.get('measurement_backend') or 'sqlite').lower() != 'sqlite' and db_loc[0] in measurement_stores): continue measurement_stores.append(db_loc[0]) tables = ['cell', 'nucleus', 'pathogen','cytoplasm'] dft, _ = _read_and_merge_data(db_loc, tables, settings['verbose'], nuclei_limit=settings['nuclei_limit'], pathogen_limit=settings['pathogen_limit']) df = pd.concat([df, dft]) _flowview_metric("objects", len(df)) _flowview_metric("databases", len(measurement_stores)) _flowview_metric("tables", len(tables) * len(measurement_stores)) try: df = calculate_shortest_distance(df, 'pathogen', 'nucleus') except Exception as e: print(e) from .training_basis import resolve_basis _basis = resolve_basis(settings) #: The column the annotation path trains against. None on the metadata #: path, where the caller's own `location_column` is the answer. Declared #: here so every branch below has it defined. _label_column = None if _basis == 'annotation': if not settings.get('annotation_column'): raise ValueError( "dataset_mode='annotation' needs annotation_column set to a " "column of png_list. Nothing else in these settings says " "which labels to train on.") _label_column = settings['annotation_column'] migrate_prediction_columns(db_loc[0]) png_list_df = _read_db(db_loc[0], tables=['png_list'])[0] if not {'prcfo', settings['annotation_column']}.issubset(png_list_df.columns): raise ValueError("The 'png_list_df' DataFrame must contain 'prcfo' and 'test' columns.") annotated_df = png_list_df[['prcfo', settings['annotation_column']]].set_index('prcfo') measurement_rows = len(df) annotation_rows = len(annotated_df) df = annotated_df.merge(df, left_index=True, right_index=True, validate='many_to_one') if df.empty: raise ValueError( f"annotation_column={settings['annotation_column']!r} joined " f"to 0 measured objects by 'prcfo' ({annotation_rows} " f"annotation rows; {measurement_rows} measurement rows), so " f"there is no training data. Verify that png_list and the " f"measurement tables come from the same source and use the " f"same object identities.") unique_values = df[settings['annotation_column']].dropna().unique() print(f"Unique values in annotation column: {unique_values}") if len(unique_values) < 2: labelled_rows = int( df[settings['annotation_column']].notna().sum()) if not len(unique_values): state = (f"has 0 non-empty labels across {len(df)} joined " f"object rows") else: state = (f"has only one observed class across " f"{labelled_rows} labelled object rows") raise ValueError( f"annotation_column={settings['annotation_column']!r} " f"{state}; binary ML training requires two real annotated " f"classes. Annotate objects in a second class, or choose the " f"annotation column that already contains both classes. " f"Unannotated objects will be scored after training; spaCR " f"will not assign them a training label.") if settings['positive_control_id'] is None and settings['negative_control_id'] is None: settings['positive_control_id'] = str(unique_values[0]) settings['negative_control_id'] = str(unique_values[1]) print(f"Automatically set positive control to {settings['positive_control_id']} and negative control to {settings['negative_control_id']} based on unique values in annotation column.") _flowview_advance("dataset") from .utils import feature_selection recruitment_channel = feature_selection(settings['channel_of_interest']) if isinstance(recruitment_channel, int): pathogen_col = f"pathogen_channel_{recruitment_channel}_mean_intensity" cytoplasm_col = f"cytoplasm_channel_{recruitment_channel}_mean_intensity" if pathogen_col in df.columns and cytoplasm_col in df.columns: df['recruitment'] = df[pathogen_col]/df[cytoplasm_col] from .batch_correction import correction_kwargs batch_kwargs = correction_kwargs( settings, default_control_column=(_label_column or settings.get('location_column')), default_control_values=settings.get('negative_control_id'), ) batch_kwargs['batch_covariate_column'] = settings.get( 'batch_covariate_column') batch_kwargs['batch_combat_mean_only'] = bool( settings.get('batch_combat_mean_only', False)) _training_column = _label_column or settings['location_column'] output, figs = ml_analysis(df, settings['channel_of_interest'], _training_column, settings['positive_control_id'], settings['negative_control_id'], settings['exclude'], settings['n_repeats'], settings['top_features'], settings['reg_alpha'], settings['reg_lambda'], settings['learning_rate'], settings['n_estimators'], settings['test_size'], settings['model_type_ml'], settings['n_jobs'], settings['remove_low_variance_features'], settings['remove_highly_correlated_features'], settings['prune_features'], settings['cross_validation'], settings['verbose'], split_by=settings.get('cv_group_by', 'well'), holdout_plate=settings.get('holdout_plate'), **batch_kwargs) shap_fig = shap_analysis(output[3], output[4], output[5]) features = output[0].select_dtypes(include=[np.number]).columns.tolist() train_features_df = pd.DataFrame(output[9], columns=['feature']) if not settings['heatmap_feature'] in features: raise ValueError(f"Variable {settings['heatmap_feature']} not found in the dataframe. Please choose one of the following: {features}") plate_heatmap = plot_plates(df=output[0], variable=settings['heatmap_feature'], grouping=settings['grouping'], min_max=settings['min_max'], cmap=settings['cmap'], min_count=settings['min_cells_per_well'], verbose=settings['verbose']) data_path, permutation_path, feature_importance_path, model_metricks_path, permutation_fig_path, feature_importance_fig_path, shap_fig_path, plate_heatmap_path, settings_csv, ml_features = get_ml_results_paths(src1, settings['model_type_ml'], settings['channel_of_interest']) df, permutation_df, feature_importance_df, _, _, _, _, _, metrics_df, _ = output _flowview_metric("objects", len(output[0])) _flowview_metric("test_objects", len(output[5])) _flowview_advance("scores") df.to_csv(data_path, mode='w', encoding='utf-8') permutation_df.to_csv(permutation_path, mode='w', encoding='utf-8') feature_importance_df.to_csv(feature_importance_path, mode='w', encoding='utf-8') train_features_df.to_csv(ml_features, mode='w', encoding='utf-8') metrics_df.to_csv(model_metricks_path, mode='w', encoding='utf-8') from .figure_sink import publish plate_heatmap_path = publish(plate_heatmap, plate_heatmap_path) permutation_fig_path = publish(figs[0], permutation_fig_path) feature_importance_fig_path = publish( figs[1], feature_importance_fig_path) shap_fig_path = write_plot(shap_fig, shap_fig_path, "SHAP summary") settings['csv_path'] = data_path settings['db_path'] = measurement_stores[0] settings['table_name'] = 'png_list' settings['update_column'] = ML_CLASS_COLUMN settings['match_column'] = 'prcfo' matched_objects = 0 unmatched_objects = 0 for store in measurement_stores: report = merge_ml_predictions( df, store, table=settings['table_name'], ) if report is not None: matched_objects += report.matched_rows unmatched_objects += report.unmatched_db_rows _flowview_metric("objects", len(df)) _flowview_metric("matched_objects", matched_objects) _flowview_metric("unmatched_objects", unmatched_objects) _flowview_metric("databases", len(measurement_stores)) return [output, plate_heatmap]
def _resolve_controls(df, location_column, negative_control, positive_control, matches): """The control values to match, and whether they had to be derived. :param matches: ``(series, control) -> boolean mask``. Passed in rather than imported because the matcher is defined inside `ml_analysis`; taking it as an argument keeps this function module-level and testable on its own. :returns: ``(negative, positive, derived)``. ``derived`` is True when the named controls matched nothing and the column's own two classes were used instead. THE CASE THIS EXISTS FOR: annotation mode points `location_column` at the annotation column, whose values are class labels, while the control settings still hold plate column names from the metadata path. Neither matches, and the user is told to "set positive_control and negative_control to values that appear there" -- for a column that already says, unambiguously, what its two classes are. Nothing is derived when the named controls DO match: an explicit choice is always honoured, including a deliberate two-of-five subset. """ if location_column not in df.columns: return negative_control, positive_control, False column = df[location_column] if isinstance(column, pd.DataFrame): return negative_control, positive_control, False any_found = (matches(column, negative_control).any() or matches(column, positive_control).any()) if any_found: return negative_control, positive_control, False present = sorted(v for v in column.dropna().unique()) if len(present) != 2: return negative_control, positive_control, False low, high = present print(f"{location_column!r} holds exactly two classes, {low!r} and " f"{high!r}, and neither {negative_control!r} nor " f"{positive_control!r} appears in it. Training on the column's own " f"classes: negative={low!r}, positive={high!r}.") return low, high, True @single_threaded_openmp('classical ML training')
[docs] def ml_analysis( df, channel_of_interest=3, location_column='columnID', positive_control='c2', negative_control='c1', exclude=None, n_repeats=10, top_features=30, reg_alpha=0.1, reg_lambda=1.0, learning_rate=0.00001, n_estimators=1000, test_size=0.2, model_type='xgboost', n_jobs=-1, remove_low_variance_features=True, remove_highly_correlated_features=True, prune_features=False, cross_validation=False, verbose=False, *, split_by='well', holdout_plate=None, batch_correction='none', batch_column='plateID', batch_control_column=None, batch_control_values=None, batch_covariate_column=None, batch_combat_mean_only=False, batch_min_samples=3, batch_missing_control='error', ): """Train a per-object classifier on positive/negative control wells and score every row of the input DataFrame. Called directly for one-off ML work, and internally by :func:`generate_ml_scores`. Filters features by channel, drops low-variance and highly correlated columns, splits (or CVs) train / test, fits the requested model, computes permutation and native feature importances, tunes an optimal decision threshold and writes predictions + probabilities back onto the returned DataFrame. :param df: Per-object feature DataFrame as produced by merging the cell/nucleus/pathogen/cytoplasm tables of a :func:`spacr.measure.measure_crop` database. :param channel_of_interest: Channel index used to select features. :param location_column: Column identifying wells / plate columns. Default ``'columnID'``. :param positive_control: Value(s) in ``location_column`` treated as the positive class. Default ``'c2'``. :param negative_control: Value(s) treated as the negative class. Default ``'c1'``. :param exclude: Columns to remove from feature space. :param n_repeats: Repeats for permutation importance. Default ``10``. :param top_features: Feature cap when ``prune_features=True``. :param reg_alpha: XGBoost L1 penalty. :param reg_lambda: XGBoost L2 penalty. :param learning_rate: XGBoost learning rate. :param n_estimators: Tree count for tree-based models. :param test_size: Test-split fraction. Default ``0.2``. :param model_type: ``'random_forest'``, ``'logistic_regression'``, ``'gradient_boosting'`` or ``'xgboost'``. :param n_jobs: Parallel job count where applicable. Default ``-1``. :param remove_low_variance_features: Drop low-variance features. :param remove_highly_correlated_features: Drop highly correlated features. :param prune_features: If True, apply ``SelectKBest`` before training. :param cross_validation: If True, run 5-fold stratified CV. :param verbose: Log progress details. :param split_by: Independent acquisition unit for train/test splitting: ``'cell'``, ``'field'``, ``'well'`` (default), or ``'plate'``. Legacy ``'none'`` is an alias for ``'cell'``. :param batch_correction: plate correction method from :mod:`spacr.batch_correction`. :param batch_column: metadata column identifying plates/batches. :param batch_control_column: metadata column holding reference-control labels for ``control_center``. :param batch_control_values: negative/reference control value(s). :param batch_min_samples: minimum rows or controls per plate. :param batch_covariate_column: Metadata column containing a biological covariate that ComBat must preserve, such as treatment, cell line, or time point. Required when ``batch_correction="combat"``; its coefficients remain in the corrected data while estimated batch effects are removed. :param batch_combat_mean_only: If ``True``, ComBat adjusts batch means without scaling batch variances. This can be appropriate when batches differ primarily by location or contain too few observations for stable variance estimates. Default ``False`` adjusts both means and variances. :param batch_missing_control: ``error`` or ``skip`` for missing controls. :returns: Tuple ``(output, figs)`` where ``output`` is a positional tuple of ``(scored_df, permutation_df, feature_importance_df, model, X_train, X_test, y_train, y_test, metrics_df, train_features)`` and ``figs`` is ``(permutation_fig, feature_importance_fig)``. :raises ValueError: on unsupported ``model_type`` or when positive / negative control rows cannot be located in ``location_column``. Example: .. code-block:: python from spacr.ml import ml_analysis output, figs = ml_analysis( df, channel_of_interest=3, positive_control='c2', negative_control='c1', model_type='xgboost', ) scored_df = output[0] See Also: :func:`generate_ml_scores` — wraps this call with DB I/O. """ from .resource_log import _guard_workers, _table_nbytes n_jobs = _guard_workers('ml_analyze', n_jobs, _table_nbytes(df)) _flowview_advance("dataset") def _match_control_values(series, control): """ Return a boolean mask selecting rows in `series` that match `control`. Matching is attempted in this order: 1. exact value match 2. numeric coercion match 3. stripped string match `control` can be a scalar or a list/tuple/set of values. """ if isinstance(control, (list, tuple, set, np.ndarray, pd.Series)): controls = list(control) else: controls = [control] mask = pd.Series(False, index=series.index) for c in controls: current_mask = pd.Series(False, index=series.index) try: current_mask |= (series == c) except Exception: pass try: s_num = pd.to_numeric(series, errors='coerce') c_num = pd.to_numeric(pd.Series([c]), errors='coerce').iloc[0] if pd.notna(c_num): current_mask |= (s_num == c_num) except Exception: pass try: s_str = series.astype(str).str.strip() c_str = str(c).strip() current_mask |= (s_str == c_str) except Exception: pass mask |= current_mask return mask from .utils import filter_dataframe_features from .plot import plot_permutation, plot_feature_importance random_state = _run_random_state(42) if 'cells_per_well' in df.columns: df = df.drop(columns=['cells_per_well']) correction_metadata = df.copy() if location_column not in df.columns: available = ", ".join(repr(c) for c in list(df.columns)[:12]) if len(df.columns) > 12: available += f", ... ({len(df.columns)} columns)" raise ValueError( f"location_column={location_column!r} is not a column of the " f"measurement table, so there is nothing to group the controls " f"by.\n The table has: {available}" f"\n If you have run this module in annotation mode, that is " f"the likely cause: versions before 1.5.0.5 wrote " f"annotation_column into location_column and never put it back, " f"so a later metadata run looked for an annotation column in the " f"measurement table. Set location_column back to your well " f"column ('columnID' or 'rowID').") if df.empty: raise ValueError( "the measurement table contains 0 object rows, so there is " "nothing to train on. Check that the selected source contains " "measured objects before running the analysis.") location_values = df[location_column] if isinstance(location_values, pd.Series): non_empty_values = location_values.dropna().astype(str).str.strip() if not non_empty_values.ne("").any(): raise ValueError( f"location_column={location_column!r} has 0 non-empty values " f"across {len(df)} object rows. Populate it with two real " f"class labels before running the analysis.") df_metadata = df[[location_column]].copy() df, features = filter_dataframe_features(df, channel_of_interest, exclude, remove_low_variance_features, remove_highly_correlated_features, verbose) print('After filtration:', len(df)) if str(batch_correction or 'none').strip().lower() not in { 'none', 'off', 'false', }: from .batch_correction import correct_from_metadata corrected, correction_report = correct_from_metadata( df[features], correction_metadata.loc[df.index], batch_correction=batch_correction, batch_column=batch_column, batch_control_column=batch_control_column, batch_control_values=batch_control_values, batch_covariate_column=batch_covariate_column, batch_combat_mean_only=batch_combat_mean_only, batch_min_samples=batch_min_samples, batch_missing_control=batch_missing_control, ) df.loc[:, features] = corrected print( f"Batch correction {correction_report.method}: " f"{correction_report.centroid_spread_before} -> " f"{correction_report.centroid_spread_after} centroid spread.") for note in correction_report.warnings: print(f"Warning: batch correction: {note}") if verbose: print(f'Found {len(features)} numerical features in the dataframe') print(f'Features used in training: {features}') print(f'Features: {features}') df = pd.concat([df, df_metadata[location_column]], axis=1) df['prcfo'] = df.index.astype(str) negative_control, positive_control, _derived_classes = _resolve_controls( df, location_column, negative_control, positive_control, _match_control_values) df1 = df[_match_control_values(df[location_column], negative_control)].copy() if verbose: print(f'Negative control: {negative_control}, samples: {len(df1)}') df2 = df[_match_control_values(df[location_column], positive_control)].copy() if verbose: print(f'Positive control: {positive_control}, samples: {len(df2)}') df1['target'] = 0 df2['target'] = 1 combined_df = pd.concat([df1, df2]) combined_df = combined_df.drop(columns=[location_column]) if verbose: print(f'Found {len(df1)} samples for {negative_control} and {len(df2)} samples for {positive_control}. Total: {len(combined_df)}') untrained = [] try: column = df[location_column] if location_column in df.columns else None if isinstance(column, pd.Series): trained_on = set(df1[location_column].unique()) | set( df2[location_column].unique()) present = set(column.dropna().unique()) untrained = sorted(str(value) for value in present - trained_on) except Exception: # noqa: BLE001 LOG.debug("could not list the classes outside the training set", exc_info=True) if untrained: print(f"{len(untrained)} class(es) of {location_column!r} are not in " f"the training set and are SCORED by a model that never saw " f"them: {untrained[:10]}" f"{'...' if len(untrained) > 10 else ''}. This fit is binary: " f"one arm is negative_control_id={negative_control!r} and the " f"other positive_control_id={positive_control!r}. Both take a " f"list, so name several values to pool them into one arm.") if df1.empty or df2.empty: column = df[location_column] if isinstance(column, pd.DataFrame): raise ValueError( f"the measurement table has {column.shape[1]} columns named " f"{location_column!r}, so the controls cannot be matched " f"against it. Drop or rename the duplicate before running " f"the analysis.") present = column.astype(str).str.strip().unique().tolist() shown = ", ".join(repr(v) for v in sorted(present)[:15]) if len(present) > 15: shown += f", ... ({len(present)} distinct values)" missing = [] if df1.empty: missing.append(f"negative_control_id={negative_control!r}") if df2.empty: missing.append(f"positive_control_id={positive_control!r}") raise ValueError( f"no rows matched {' and '.join(missing)} in column " f"{location_column!r}, so there is nothing to train on.\n" f" {location_column!r} contains: {shown}\n" f" Set positive_control_id and negative_control_id to values " f"that appear there, or set location_column to the column that " f"holds your controls.") X = combined_df[features] y = combined_df['target'] if prune_features: before_pruning = len(X.columns) selector = SelectKBest(score_func=f_classif, k=top_features) X_selected = selector.fit_transform(X, y) selected_features = X.columns[selector.get_support()] X = pd.DataFrame(X_selected, columns=selected_features, index=X.index) features = selected_features.tolist() after_pruning = len(X.columns) print(f"Removed {before_pruning - after_pruning} features using SelectKBest") _flowview_metric("objects", len(df)) _flowview_metric("training_objects", len(combined_df)) _flowview_metric("features", len(features)) _flowview_advance("split") from .classifier_evaluation import grouped_split, split_group_values split_frame = combined_df[['prcfo']].reset_index(drop=True) split_level, split_groups = split_group_values( group_by=split_by, frame=split_frame, table='ML control measurements') held = holdout_plate if held is not None and not isinstance(held, (list, tuple, set)): held = [held] if held: _plate_level, plate_groups = split_group_values( group_by='plate', frame=split_frame, table='ML control measurements') train_index, test_index, split_report = grouped_split( plate_groups, y.to_numpy(), test_size, seed=random_state, group_by='plate', hold_out_groups=held) else: train_index, test_index, split_report = grouped_split( split_groups, y.to_numpy(), test_size, seed=random_state, group_by=split_level) X_train, X_test = X.iloc[train_index], X.iloc[test_index] y_train, y_test = y.iloc[train_index], y.iloc[test_index] print(split_report.summary()) combined_df['data_usage'] = 'train' combined_df.loc[X_test.index, 'data_usage'] = 'test' df['data_usage'] = 'not_used' df.loc[combined_df.index, 'data_usage'] = combined_df['data_usage'] df['data_usage_group_by'] = split_report.group_by df['split_requested_fraction'] = split_report.requested_fraction df['split_cell_fraction'] = split_report.cell_fraction df['split_group_fraction'] = split_report.group_fraction _flowview_metric("objects", len(X)) _flowview_metric("train_objects", len(X_train)) _flowview_metric("test_objects", len(X_test)) _flowview_advance("model") if model_type == 'random_forest': model = RandomForestClassifier(n_estimators=n_estimators, random_state=random_state, n_jobs=n_jobs) elif model_type == 'extra_trees': from sklearn.ensemble import ExtraTreesClassifier model = ExtraTreesClassifier(n_estimators=n_estimators, random_state=random_state, n_jobs=n_jobs) elif model_type == 'logistic_regression': model = LogisticRegression(max_iter=1000, random_state=random_state) elif model_type == 'gradient_boosting': model = HistGradientBoostingClassifier(max_iter=n_estimators, random_state=random_state) elif model_type == 'xgboost': model = XGBClassifier( reg_alpha=reg_alpha, reg_lambda=reg_lambda, learning_rate=learning_rate, n_estimators=n_estimators, random_state=random_state, nthread=n_jobs, eval_metric='logloss', ) elif model_type == 'lightgbm': try: from lightgbm import LGBMClassifier except ImportError: raise ImportError("model_type='lightgbm' requires the 'lightgbm' package. Install it with: pip install lightgbm") model = LGBMClassifier(n_estimators=n_estimators, learning_rate=learning_rate, reg_alpha=reg_alpha, reg_lambda=reg_lambda, random_state=random_state, n_jobs=n_jobs) elif model_type == 'catboost': try: from catboost import CatBoostClassifier except ImportError: raise ImportError("model_type='catboost' requires the 'catboost' package. Install it with: pip install catboost") model = CatBoostClassifier(iterations=n_estimators, learning_rate=learning_rate, l2_leaf_reg=reg_lambda, random_state=random_state, thread_count=n_jobs, verbose=False) elif model_type == 'svm': from sklearn.calibration import CalibratedClassifierCV from sklearn.svm import SVC model = CalibratedClassifierCV( estimator=SVC(random_state=random_state), method='sigmoid', cv=3, n_jobs=n_jobs, ensemble=False, ) elif model_type == 'mlp': from sklearn.neural_network import MLPClassifier model = MLPClassifier(max_iter=max(200, n_estimators), random_state=random_state) else: raise ValueError(f"Unsupported model_type: {model_type}") model.spacr_split_report_ = split_report.to_dict() _flowview_metric("features", len(X.columns)) _flowview_advance("training") if cross_validation: from .io import make_cv_folds distinct_groups = len(np.unique(split_groups)) n_folds = min(5, distinct_groups) folds = make_cv_folds( y.to_numpy(), n_folds, groups=split_groups, seed=random_state) expected_classes = set(np.unique(y)) fold_metrics = [] for fold_idx, (train_index, test_index) in enumerate(folds, start=1): if (set(np.unique(y.iloc[train_index])) != expected_classes or set(np.unique(y.iloc[test_index])) != expected_classes): raise ValueError( f"{split_level}-grouped CV fold {fold_idx} cannot put " "every class in both train and test. Add independent " f"class-bearing {split_level}s or choose a finer split.") X_train, X_test = X.iloc[train_index], X.iloc[test_index] y_train, y_test = y.iloc[train_index], y.iloc[test_index] model.fit(X_train, y_train) predictions_test = model.predict(X_test) combined_df.loc[X_test.index, 'predictions'] = predictions_test prediction_probabilities_test = model.predict_proba(X_test) optimal_threshold = find_optimal_threshold(y_test, prediction_probabilities_test[:, 1]) if verbose: print(f'Fold {fold_idx} - Optimal threshold: {optimal_threshold}') df.loc[X_test.index, 'predictions'] = predictions_test for i in range(prediction_probabilities_test.shape[1]): df.loc[X_test.index, f'prediction_probability_class_{i}'] = prediction_probabilities_test[:, i] fold_report = classification_report( y_test, predictions_test, output_dict=True, zero_division=0) fold_metrics.append(pd.DataFrame(fold_report).transpose()) if verbose: print(f"Fold {fold_idx} Classification Report:") print(classification_report( y_test, predictions_test, zero_division=0)) metrics_df = pd.concat(fold_metrics).groupby(level=0).mean() model.fit(X, y) all_predictions = model.predict(df[features]) df['predictions'] = all_predictions prediction_probabilities = model.predict_proba(df[features]) for i in range(prediction_probabilities.shape[1]): df[f'prediction_probability_class_{i}'] = prediction_probabilities[:, i] else: model.fit(X_train, y_train) predictions_test = model.predict(X_test) combined_df.loc[X_test.index, 'predictions'] = predictions_test prediction_probabilities_test = model.predict_proba(X_test) optimal_threshold = find_optimal_threshold(y_test, prediction_probabilities_test[:, 1]) if verbose: print(f'Optimal threshold: {optimal_threshold}') X_all = df[features] all_predictions = model.predict(X_all) df['predictions'] = all_predictions prediction_probabilities = model.predict_proba(X_all) for i in range(prediction_probabilities.shape[1]): df[f'prediction_probability_class_{i}'] = prediction_probabilities[:, i] if verbose: print("\nClassification Report:") print(classification_report( y_test, predictions_test, zero_division=0)) report_dict = classification_report( y_test, predictions_test, output_dict=True, zero_division=0) metrics_df = pd.DataFrame(report_dict).transpose() _flowview_metric("objects", len(X)) _flowview_metric("features", len(features)) _flowview_advance("evaluation") metrics_df['split_group_by'] = split_report.group_by metrics_df['split_requested_fraction'] = split_report.requested_fraction metrics_df['split_group_fraction'] = split_report.group_fraction metrics_df['split_cell_fraction'] = split_report.cell_fraction perm_importance = permutation_importance(model, X_train, y_train, n_repeats=n_repeats, random_state=random_state, n_jobs=guarded_n_jobs(n_jobs, 'permutation importance')) permutation_df = pd.DataFrame({ 'feature': [features[i] for i in perm_importance.importances_mean.argsort()], 'importance_mean': perm_importance.importances_mean[perm_importance.importances_mean.argsort()], 'importance_std': perm_importance.importances_std[perm_importance.importances_mean.argsort()] }).tail(top_features) permutation_fig = plot_permutation(permutation_df) if verbose: permutation_fig.show() if hasattr(model, 'feature_importances_'): feature_importances = model.feature_importances_ feature_importance_df = pd.DataFrame({ 'feature': features, 'importance': feature_importances }).sort_values(by='importance', ascending=False).head(top_features) feature_importance_fig = plot_feature_importance(feature_importance_df) if verbose: feature_importance_fig.show() else: feature_importance_df = permutation_df.rename( columns={"importance_mean": "importance"} )[["feature", "importance"]].sort_values( by="importance", ascending=False).head(top_features) feature_importance_fig = plot_feature_importance( feature_importance_df, title=f"Top {len(feature_importance_df)} features " f"(permutation importance)") if verbose: feature_importance_fig.show() df = _calculate_similarity(df, features, location_column, positive_control, negative_control) df['prcfo'] = df.index.astype(str) df = _assign_prcfo_parts(df, object_column='object') df['prc'] = _compose_prc_column(df) return [df, permutation_df, feature_importance_df, model, X_train, X_test, y_train, y_test, metrics_df, features], [permutation_fig, feature_importance_fig]
#: How many background rows a model-agnostic explainer is given. The #: permutation explainer is O(background x features) per explained row, so #: the whole training set turns a 0.3 s panel into minutes. Summarising the #: background is what the SHAP authors recommend for exactly this. SHAP_BACKGROUND = 100 def _shap_values(model, X_train, X_test): """(values, note). SHAP contributions for every model the panel offers. THE FAILURE IS NOT ALWAYS AT CONSTRUCTION. `shap.Explainer(model, X)` accepts an xgboost booster happily and raises "Categorical split is not yet supported" only when it is CALLED -- so a fallback chosen at construction time never ran, and the panel's DEFAULT model produced no SHAP at all. Each candidate is therefore tried all the way through. """ import shap attempts = list(_shap_explainers(model, X_train)) trouble = "" for explainer, note in attempts: try: return explainer(X_test), note except Exception as error: # noqa: BLE001 trouble = f"{type(error).__name__}: {error}" LOG.debug("a SHAP explainer would not run", exc_info=True) raise RuntimeError( f"No SHAP explainer could explain {type(model).__name__}. The last " f"failure was: {trouble}") def _shap_explainers(model, X_train): """Every explainer worth trying for this estimator, best first. THREE OF THE NINE MODELS THE PANEL OFFERS COULD NOT BE EXPLAINED AT ALL, including the default: * `xgboost` raised "Categorical split is not yet supported. You can still use TreeExplainer with feature_perturbation=tree_path_dependent" -- an error carrying its own fix, which nothing acted on. * `svm` and `mlp` raised "The passed model is not callable and cannot be analyzed directly with the given masker". A support vector machine and a neural net are not trees and not linear; the model-agnostic explainer takes a FUNCTION, not an estimator. The note is returned rather than printed here so the caller decides where it goes, and it is said out loud because the three explainers do not compute the same quantity: `tree_path_dependent` conditions on the tree's own splits rather than on an independent background, and the permutation explainer estimates rather than solves. """ import shap try: automatic_note = "" if type(model).__module__.split(".", 1)[0] == "xgboost": automatic_note = ( "SHAP: XGBoost was accepted by the automatic tree " "explainer, so the panel shows tree contributions rather " "than a model-agnostic estimate." ) yield shap.Explainer(model, X_train), automatic_note except Exception: # noqa: BLE001 LOG.debug("the default SHAP explainer would not build", exc_info=True) try: yield (shap.TreeExplainer( model, feature_perturbation="tree_path_dependent"), "SHAP: this model has categorical splits, so it is explained " "with feature_perturbation='tree_path_dependent' -- which " "conditions on the tree's own splits rather than on an " "independent background.") except Exception: # noqa: BLE001 LOG.debug("this model is not a tree", exc_info=True) predict = (getattr(model, "predict_proba", None) or getattr(model, "decision_function", None) or getattr(model, "predict", None)) if predict is None: return background = X_train if hasattr(X_train, "shape") and X_train.shape[0] > SHAP_BACKGROUND: background = shap.utils.sample(X_train, SHAP_BACKGROUND, random_state=0) try: yield (shap.Explainer(predict, background), f"SHAP: {type(model).__name__} is neither a tree nor a " f"linear model, so it is explained through its predictions " f"over {len(background)} background row(s). That is an " f"ESTIMATE of each contribution rather than an exact " f"decomposition.") except Exception: # noqa: BLE001 LOG.debug("the model-agnostic SHAP explainer would not build", exc_info=True)
[docs] def shap_analysis(model, X_train, X_test): """Build a SHAP summary beeswarm for ``X_test``. The beeswarm is rendered with pyqtgraph so it can be embedded in the same scene-based figure workflow as other model-explanation plots. The function returns a live :class:`~spacr.qt.widgets.fast_plots.FastPlot`; it neither writes a file nor returns a matplotlib figure. Pass the result to :func:`write_plot` to export it in the configured figure format. :param model: Fitted estimator compatible with ``shap.Explainer``. :param X_train: Training features used to seed the explainer. :param X_test: Test features to explain. :returns: A ``FastPlot`` holding the beeswarm, or ``None`` when Qt is unavailable or the attribution matrix cannot be plotted. """ import shap shap_values, note = _shap_values(model, X_train, X_test) if note: print(note) if len(shap_values.shape) == 3: output_index = 1 if shap_values.shape[-1] > 1 else 0 shap_values = shap_values[..., output_index] from .figures.headless import application application_object, refusal = application() if application_object is None: print(refusal) return None from .qt.widgets.fast_plots import FastPlot matrix = np.asarray(shap_values.values, dtype=float) if matrix.ndim != 2 or not matrix.size: return None columns = list(X_test.columns)[:matrix.shape[1]] order = np.argsort(np.nanmean(np.abs(matrix), axis=0))[::-1] names = [str(columns[int(i)]) for i in order] plot = FastPlot(title="SHAP summary", x_label="SHAP value", y_label="") plot.resize(1200, max(420, 34 * len(names) + 140)) if not plot.add_beeswarm(names, matrix[:, order], X_test[names].to_numpy(dtype=float)): plot.deleteLater() return None application_object.processEvents() return plot
[docs] def write_plot(plot, path, title=""): """Write a pyqtgraph plot out and announce it, like ``publish`` does. The counterpart of :func:`spacr.figure_sink.publish` for a scene rather than a matplotlib figure: the format follows the user's preference, the file NAME follows the format, and the written file reaches the gallery, because saved and visible are the same event. ``None`` writes nothing, announces nothing and returns None -- a plot that could not be built must not take the run down after the model has been fitted and every object scored. :param plot: a ``FastPlot``, or None. :param path: where to write it; the extension may be rewritten. :param title: the name the gallery tile carries. :returns: the path written, or None. """ if plot is None: return None from .figure_sink import publish_file from .plot import figure_output_preferences chosen = str(figure_output_preferences()[0]).lower().lstrip('.') stem, _ = os.path.splitext(str(path)) target = f"{stem}.{chosen}" parent = os.path.dirname(os.path.abspath(target)) os.makedirs(parent, exist_ok=True) try: written = plot.export(target) finally: plot.deleteLater() if written: publish_file(written, title=title or None) return written
[docs] def find_optimal_threshold(y_true, y_pred_proba): """Return the probability threshold maximising F1 on the precision-recall curve. :param y_true: Ground-truth binary labels. :param y_pred_proba: Predicted probabilities for the positive class. :returns: Optimal probability threshold. """ precision, recall, thresholds = precision_recall_curve(y_true, y_pred_proba) denominator = precision + recall with np.errstate(divide='ignore', invalid='ignore'): f1_scores = np.where(denominator > 0, 2 * (precision * recall) / denominator, 0.0) optimal_idx = np.argmax(f1_scores) optimal_threshold = thresholds[optimal_idx] return optimal_threshold
def _calculate_similarity(df, features, col_to_compare, val1, val2): """ Calculate similarity scores of each well to the positive and negative controls using various metrics. Args: df (pandas.DataFrame): DataFrame containing the data. features (list): List of feature columns to use for similarity calculation. col_to_compare (str): Column name to use for comparing groups. val1, val2 (str): Values in col_to_compare to create subsets for comparison. Returns: pandas.DataFrame: DataFrame with similarity scores. """ if isinstance(val1, str): pos_control = df[df[col_to_compare] == val1][features].mean() elif isinstance(val1, list): pos_control = df[df[col_to_compare].isin(val1)][features].mean() if isinstance(val2, str): neg_control = df[df[col_to_compare] == val2][features].mean() elif isinstance(val2, list): neg_control = df[df[col_to_compare].isin(val2)][features].mean() scaler = StandardScaler() scaled_features = scaler.fit_transform(df[features]) cov_matrix = np.cov(scaled_features, rowvar=False) inv_cov_matrix = None try: inv_cov_matrix = np.linalg.inv(cov_matrix) except np.linalg.LinAlgError: epsilon = 1e-5 inv_cov_matrix = np.linalg.inv(cov_matrix + np.eye(cov_matrix.shape[0]) * epsilon) def safe_similarity(func, row, control, *args, **kwargs): """Call ``func(row, control, ...)`` and swallow errors (return ``NaN``).""" try: return func(row, control, *args, **kwargs) except Exception: return np.nan try: df['similarity_to_pos_euclidean'] = df[features].apply(lambda row: safe_similarity(euclidean, row, pos_control), axis=1) df['similarity_to_neg_euclidean'] = df[features].apply(lambda row: safe_similarity(euclidean, row, neg_control), axis=1) df['similarity_to_pos_cosine'] = df[features].apply(lambda row: safe_similarity(cosine, row, pos_control), axis=1) df['similarity_to_neg_cosine'] = df[features].apply(lambda row: safe_similarity(cosine, row, neg_control), axis=1) df['similarity_to_pos_mahalanobis'] = df[features].apply(lambda row: safe_similarity(mahalanobis, row, pos_control, inv_cov_matrix), axis=1) df['similarity_to_neg_mahalanobis'] = df[features].apply(lambda row: safe_similarity(mahalanobis, row, neg_control, inv_cov_matrix), axis=1) df['similarity_to_pos_manhattan'] = df[features].apply(lambda row: safe_similarity(cityblock, row, pos_control), axis=1) df['similarity_to_neg_manhattan'] = df[features].apply(lambda row: safe_similarity(cityblock, row, neg_control), axis=1) df['similarity_to_pos_minkowski'] = df[features].apply(lambda row: safe_similarity(minkowski, row, pos_control, p=3), axis=1) df['similarity_to_neg_minkowski'] = df[features].apply(lambda row: safe_similarity(minkowski, row, neg_control, p=3), axis=1) df['similarity_to_pos_chebyshev'] = df[features].apply(lambda row: safe_similarity(chebyshev, row, pos_control), axis=1) df['similarity_to_neg_chebyshev'] = df[features].apply(lambda row: safe_similarity(chebyshev, row, neg_control), axis=1) df['similarity_to_pos_braycurtis'] = df[features].apply(lambda row: safe_similarity(braycurtis, row, pos_control), axis=1) df['similarity_to_neg_braycurtis'] = df[features].apply(lambda row: safe_similarity(braycurtis, row, neg_control), axis=1) except Exception as e: print(f"Error calculating similarity scores: {e}") return df def _announce_the_bundle(folder, title): """Put ONE tile in the gallery for a bundle's figure. ONE PICTURE, ONE TILE. A bundle holds the same figure twice, as a PDF and as a PNG, because a folder somebody opens should carry both -- but announcing both puts two tiles in the gallery for one picture, and a reader clicking each of them to find out they are the same is exactly the confusion the gallery exists to remove. The one announced is the one in the format the user chose. :param folder: the bundle directory. :param title: the name the tile carries. :returns: the path announced, or None when the folder holds no figure. """ from .figure_sink import publish_file from .plot import figure_output_preferences if not folder or not os.path.isdir(folder): return None wanted = str(figure_output_preferences()[0]).lower().lstrip('.') written = sorted(os.listdir(folder)) chosen = next((f for f in written if f.lower().endswith(f".{wanted}")), None) if chosen is None: chosen = next((f for f in written if f.lower().endswith(".pdf")), None) if chosen is None: return None return publish_file(os.path.join(folder, chosen), title) _EPHEMERAL_FIGURES = None def _figure_folder(src, save): """Where a drawn figure goes: the run folder, or a temporary one. `save` GATES THE RUN FOLDER, NOT THE PICTURE. Before these charts moved to pyqtgraph they were `plt.show()`\\ n and never written, so a `save=False` run still SAW them -- and writing them into the user's results folder now would be a behaviour change nobody asked for. A temporary directory is what an ephemeral figure has always been; the gallery gets its tile either way, because saved and visible are one event. :param src: plate folder. :param save: whether this run is writing its results. :returns: a directory that exists. """ global _EPHEMERAL_FIGURES if save: folder = os.path.join(str(src), 'results') os.makedirs(folder, exist_ok=True) return folder if _EPHEMERAL_FIGURES is None: import tempfile _EPHEMERAL_FIGURES = tempfile.mkdtemp(prefix="spacr-figures-") return _EPHEMERAL_FIGURES def _draw_response_panel_in_pyqtgraph(values, transform, column, src): """Draw the response distribution before and after, and publish it. :param values: the untransformed response. :param transform: the transformation named in the settings. :param column: the response's own column name. :param src: plate folder, or None to draw without writing a file. :returns: the path written, or None. """ from .figures.headless import application application_object, refusal = application() if application_object is None: print(refusal) return None from .response_distribution import fast_panel plot = fast_panel(values, transform, dependent_variable=column) if plot is None: print("the response distribution panel was not drawn: the " "response holds no finite values") return None plot.resize(1100, 660) application_object.processEvents() if not src: plot.deleteLater() return None return write_plot( plot, os.path.join(_figure_folder(src, True), 'response_distribution.pdf'), "Response distribution") def _draw_shap_summary_in_pyqtgraph(shap_values, sample, src, name, top, save=True): """Draw a SHAP beeswarm in pyqtgraph and write its bundle. The features are ranked by MEAN ABSOLUTE contribution, which is the order `shap.summary_plot` uses and the only one that answers "which of these matters": a feature that pushes hard in both directions has a mean near zero and belongs at the top, not the bottom. :param shap_values: a shap Explanation, or anything with ``.values``. :param sample: the frame the values were computed over. :param src: plate folder; the bundle goes under ``<src>/results``. :param name: bundle name. :param top: how many features to show. :returns: the folder written, or None when there is no Qt. """ from .figures.headless import application from .figure_sink import publish_file application_object, refusal = application() if application_object is None: print(refusal) return None from .qt.widgets.fast_plots import FastPlot from .figures.bundle import save as write_bundle matrix = np.asarray(getattr(shap_values, 'values', shap_values), dtype=float) if matrix.ndim != 2 or not matrix.size: return None columns = list(sample.columns)[:matrix.shape[1]] strength = np.nanmean(np.abs(matrix), axis=0) order = np.argsort(strength)[::-1][:int(top)] names = [str(columns[int(i)]) for i in order] picked = matrix[:, order] values = sample[names].to_numpy(dtype=float) title = f"SHAP summary - top {len(names)} features" plot = FastPlot(title=title, x_label="SHAP value", y_label="") try: plot.resize(1200, max(420, 34 * len(names) + 140)) if not plot.add_beeswarm(names, picked, values): return None application_object.processEvents() folder = write_bundle(_figure_folder(src, save), name, render=plot.export, data=plot.beeswarm_frame(), groups=None, unit="observation", settings={"top_features": int(top), "figure": name}) finally: plot.deleteLater() _announce_the_bundle(folder, title) return folder def _draw_the_cell_count_sweep(summary, mark, path): """Draw the sample-size sweep in pyqtgraph and write it out. PUBLISHED, NOT SHOWN. `plt.show()` here blocked forever anywhere there was no GUI event loop to hand it to: with the Qt backend it calls `start_main_loop`, and a script, a notebook or `spacr-run regression` then sat in `qt_compat._exec` until it was killed. Saved and visible are ONE event, through the figure sink, and a figure reaching the gallery must not depend on somebody calling `show`. :param summary: frame with ``sample_size``, ``smoothed_mean_abs_diff`` and ``std_abs_diff``. :param mark: the sample size the threshold line is drawn at. :param path: destination; the extension follows the format preference. :returns: the path written, or None when there is no Qt to draw under. """ from .figures.headless import application application_object, refusal = application() if application_object is None: print(refusal) return None from .qt.widgets.fast_plots import FastPlot from .figures.style import ROLES sizes = summary['sample_size'].to_numpy(dtype=float) middle = summary['smoothed_mean_abs_diff'].to_numpy(dtype=float) spread = summary['std_abs_diff'].to_numpy(dtype=float) plot = FastPlot(title="Mean absolute difference against sample size", x_label="Sample size", y_label="Mean absolute difference") try: plot.resize(1100, 760) if not plot.add_curve(sizes, middle, low=middle - spread, high=middle + spread): return None plot.add_line(x=float(mark), colour=ROLES["reference"], label="minimum cell count") application_object.processEvents() return write_plot(plot, path, "Minimum cell count") except Exception: # noqa: BLE001 plot.deleteLater() raise def _figure_name_for(title): """A filename from a figure title: lower case, words joined by _.""" keep = [ch.lower() if ch.isalnum() else " " for ch in str(title)] return "_".join("".join(keep).split()) or "figure" def _draw_radar_in_pyqtgraph(labels, values, title, src, name, save=True): """Draw a radar in pyqtgraph and write its bundle under ``<src>/results``. Returns the folder, or None when there is no Qt to render under -- which `render`'s own refusal explains rather than leaving the run silent. """ from .figures.headless import application from .figure_sink import publish_file application_object, refusal = application() if application_object is None: print(refusal) return None from .qt.widgets.fast_plots import FastPlot from .figures.bundle import save as write_bundle plot = FastPlot(title=title, x_label="", y_label="") try: plot.resize(820, 780) if not plot.add_radar(labels, values): return None application_object.processEvents() folder = write_bundle(_figure_folder(src, save), name, render=plot.export, data=plot.radar_frame(), groups=None, unit="feature", settings={"figure": name}) finally: plot.deleteLater() _announce_the_bundle(folder, title) return folder def _draw_importance_in_pyqtgraph(frame, title, src, name, top, save=True): """Draw a ranked importance chart in pyqtgraph and write it out. The regression and explanation figures were drawn twice -- once in pyqtgraph for the tab and once in matplotlib for the file -- so one screen produced two pictures of one number from two code paths that can disagree. These two were the last that could not move, because twenty feature names need HORIZONTAL bars and the plot could not draw them. Returns the bundle folder, or None when there is no Qt to render under. A None is not a failure: the caller has already written the CSV, and `render_bundle` says out loud why it could not draw. :param frame: importance table with ``feature`` and ``importance``. :param title: the figure's title. :param src: plate folder; the bundle goes under ``<src>/results``. :param name: bundle name. :param top: how many features to show. :returns: the folder written, or None. """ from .figures.headless import application from .figure_sink import publish_file application_object, refusal = application() if application_object is None: print(refusal) return None from .qt.widgets.fast_plots import FastPlot from .figures.bundle import save as write_bundle shown = frame.head(int(top)) plot = FastPlot(title=title, x_label="Importance", y_label="") try: plot.resize(1200, max(420, 34 * len(shown) + 140)) if not plot.add_ranked_bars(list(shown['feature']), list(shown['importance']), highlight=3, descending=False): return None application_object.processEvents() folder = write_bundle(_figure_folder(src, save), name, render=plot.export, data=plot.ranked_frame(), groups=None, unit="feature", settings={"top_features": int(top), "figure": name}) finally: plot.deleteLater() _announce_the_bundle(folder, title) return folder def _save_importance_csv(df, src, filename): """Write an importance table to ``<src>/results/<filename>``. :param df: Importance DataFrame with ``feature`` / ``importance``. :param src: Plate folder the explained model was scored from. :param filename: Basename of the CSV to write. :returns: The full path written. """ results_loc = os.path.join(src, 'results') os.makedirs(results_loc, exist_ok=True) out_path = os.path.join(results_loc, filename) df.to_csv(out_path, index=False) print(f"Saved {out_path}") return out_path
[docs] def interpret_vision_model(settings=None): """Explain a spacr vision-model score using RF, permutation and SHAP importance, with per-compartment / per-channel radar plots. Merges per-object measurements from the selected measurement store with a CSV of predicted scores, runs any combination of RF feature importance, permutation importance and SHAP over the top features, then aggregates SHAP contributions into compartment and channel radar plots so you can see which region (cell / nucleus / pathogen / cytoplasm) and which fluorescence channel drives the model. :param settings: Settings dict, canonicalized via :func:`spacr.settings.set_interpret_vision_model_defaults`. Key entries: - ``src`` — folder containing the measurements. - ``measurement_backend`` / ``measurement_backend_target`` — select the SQLite, DuckDB, Parquet or PostgreSQL measurement store. - ``scores`` — CSV of per-object predictions to explain. - ``score_column`` — column of ``scores`` holding the score. - ``tables`` — DB tables to merge (default ``['cell','nucleus','pathogen','cytoplasm']``). - ``feature_importance`` / ``permutation_importance`` / ``shap`` — enable each explainer. - ``top_features`` — cap on features shown. - ``nuclei_limit`` / ``pathogen_limit`` — object-count caps. - ``n_jobs``, ``save``. :returns: The merged per-object DataFrame — the measurement tables joined to the scores CSV — that the explainers were fitted on. Radar and importance plots are rendered, and with ``save=True`` importance CSVs are written under the source folder's ``results/``, as side effects. Example: .. code-block:: python from spacr.ml import interpret_vision_model interpret_vision_model({ 'src': '/data/plate01', 'scores': '/data/plate01/results/pred.csv', 'score_column': 'pred', 'shap': True, 'top_features': 30, }) See Also: :func:`spacr.submodules.interpret_vision_model` — legacy / alternative entry point returning a dict of importance DataFrames instead of the merged measurements. """ if settings is None: settings = {} from .io import (_read_and_merge_data, _report_fan_out, JoinFanOut, TimelapseKeyMismatch) from .predictions import crop_name_metadata from .settings import set_interpret_vision_model_defaults from .utils import save_settings, _time_column, _measurement_store_for settings = set_interpret_vision_model_defaults(settings) save_settings(settings, name='interperate_vision_model', show=True) def create_extended_radar_plot(values, labels, title): """Draw a filled radar for ``values`` labelled by ``labels``. A RADAR IS A POLYGON, NOT AN AXIS. This was the last figure on the explanation path that could not move to the screen's renderer, on the grounds that pyqtgraph has no polar view -- which is true and was never the obstacle: each label takes an angle, each value a radius, and `FastPlot.add_radar` draws its own rings because a radar read against a square grid is unreadable. """ return _draw_radar_in_pyqtgraph( list(labels), list(values), title, settings['src'], _figure_name_for(title), settings['save']) def extract_compartment_channel(feature_name): """Return ``(compartment, channel)`` parsed from a feature column name.""" compartment = feature_name.split('_')[0] if compartment == 'cells': compartment = 'cell' channels = [] if 'channel_0' in feature_name: channels.append('channel_0') if 'channel_1' in feature_name: channels.append('channel_1') if 'channel_2' in feature_name: channels.append('channel_2') if 'channel_3' in feature_name: channels.append('channel_3') if channels: channel = ' + '.join(channels) else: channel = 'morphology' return (compartment, channel) def read_and_preprocess_data(settings): """Merge measurement DB tables with a scores CSV and split into ``(X, y, merged_df)``.""" sqlite_path = os.path.join(settings['src'], 'measurements', 'measurements.db') measurement_store = _measurement_store_for(sqlite_path, settings) or sqlite_path df, _ = _read_and_merge_data( locs=[measurement_store], tables=settings['tables'], verbose=True, nuclei_limit=settings['nuclei_limit'], pathogen_limit=settings['pathogen_limit'] ) scores_df = tabular.read_table(settings['scores']) df['object_label'] = df['object_label'].str.replace('o', '') join_cols = ['plateID', 'rowID', 'columnID', 'fieldID', 'object_label'] df_time = _time_column(df.columns) name_col = next((c for c in ('path', 'png_path', 'file_name') if c in scores_df.columns), None) if name_col is not None: parsed = crop_name_metadata(scores_df[name_col], timelapse=df_time is not None) for col in parsed.columns: if col != 'prcfo': scores_df[col] = parsed[col] if 'object_label' not in scores_df.columns: scores_df['object_label'] = scores_df['object'] df['object_label'] = df['object_label'].str.replace('o', '').astype(str) scores_time = _time_column(scores_df.columns) if df_time is not None and scores_time is not None: if df_time != scores_time: scores_df = scores_df.rename(columns={scores_time: df_time}) join_cols = join_cols + [df_time] elif df_time is not None or scores_time is not None: raise TimelapseKeyMismatch( f"{settings['scores']} and the measurements database disagree " f"about the timepoint: the scores have {scores_time!r} and the " f"objects have {df_time!r}. One of the two was produced by a " f"non-timelapse run, so there is no timepoint to join on, and " f"joining without it would match every frame's object to every " f"frame's score. Re-score the dataset, or supply a scores file " f"that carries the crop file name so the timepoint can be read " f"off it.") df[join_cols] = df[join_cols].astype(str) scores_df[join_cols] = scores_df[join_cols].astype(str) scores_df = scores_df[join_cols + [settings['score_column']]] try: merged_df = pd.merge(df, scores_df, on=join_cols, how='inner', validate='many_to_one') except pd.errors.MergeError as error: duplicated = scores_df[scores_df.duplicated(subset=join_cols, keep=False)] examples = (duplicated[join_cols].drop_duplicates() .head(3).to_dict('records')) raise JoinFanOut( f"{settings['scores']} holds more than one score for the same " f"object: {list(join_cols)} repeats " f"{len(duplicated[join_cols].drop_duplicates())} time(s), e.g. " f"{examples}. Joining it to the measurements would put those " f"objects into the training set once per duplicate row, so " f"every measurement in the result is duplicated. This usually " f"means the scoring step ran twice and appended a second set " f"of rows; de-duplicate the scores file before reading it." ) from error _report_fan_out(df, merged_df, join_cols, left_name='object', right_name='scores') X = schema.model_feature_frame( merged_df, exclude=[settings['score_column']], ) y = merged_df[settings['score_column']] return X, y, merged_df X, y, merged_df = read_and_preprocess_data(settings) if settings['feature_importance'] or settings['permutation_importance'] or settings['shap']: model = RandomForestClassifier(random_state=_run_random_state(42), n_jobs=settings['n_jobs']) model.fit(X, y) feature_importances = model.feature_importances_ feature_importance_df = pd.DataFrame({'feature': X.columns, 'importance': feature_importances}) feature_importance_df = feature_importance_df.sort_values(by='importance', ascending=False) if settings['feature_importance']: print(f"Feature Importance ...") top_feature_importance_df = feature_importance_df.head(settings['top_features']) _draw_importance_in_pyqtgraph( feature_importance_df, f"Top {settings['top_features']} Features - Feature " f"Importance", settings['src'], 'feature_importance', settings['top_features'], settings['save']) if settings['save']: _save_importance_csv(feature_importance_df, settings['src'], 'feature_importance.csv') if settings['permutation_importance']: print(f"Permutation Importance ...") perm_importance = permutation_importance(model, X, y, n_repeats=10, random_state=_run_random_state(42), n_jobs=settings['n_jobs']) perm_importance_df = pd.DataFrame({'feature': X.columns, 'importance': perm_importance.importances_mean}) perm_importance_df = perm_importance_df.sort_values(by='importance', ascending=False) top_perm_importance_df = perm_importance_df.head(settings['top_features']) _draw_importance_in_pyqtgraph( perm_importance_df, f"Top {settings['top_features']} Features - Permutation " f"Importance", settings['src'], 'permutation_importance', settings['top_features'], settings['save']) if settings['save']: _save_importance_csv(perm_importance_df, settings['src'], 'permutation_importance.csv') if settings['shap']: import shap print(f"SHAP Analysis ...") top_features = feature_importance_df.head(settings['top_features'])['feature'] X_top = X[top_features] model = RandomForestClassifier(random_state=_run_random_state(42), n_jobs=settings['n_jobs']) model.fit(X_top, y) if settings['shap_sample']: sample = max(1, min(int(len(X_top) / 100), len(X_top))) X_sample = X_top.sample(sample, random_state=_run_random_state(42)) else: X_sample = X_top explainer = shap.Explainer(model.predict, X_sample) shap_values = explainer(X_sample, max_evals=1500) _draw_shap_summary_in_pyqtgraph( shap_values, X_sample, settings['src'], 'shap_summary', settings['top_features'], settings['save']) shap_df = pd.DataFrame(shap_values.values, columns=X_sample.columns) shap_df.columns = pd.MultiIndex.from_tuples( [extract_compartment_channel(feat) for feat in shap_df.columns], names=['compartment', 'channel'] ) shap_features = shap_df.abs().T compartment_mean = ( shap_features.groupby(level='compartment').mean().mean(axis=1)) channel_mean = ( shap_features.groupby(level='channel').mean().mean(axis=1)) combined_compartment = {} for i, comp1 in enumerate(compartment_mean.index): for comp2 in compartment_mean.index[i+1:]: combined_compartment[f"{comp1} + {comp2}"] = shap_df.loc[:, (comp1, slice(None))].abs().mean().mean() + \ shap_df.loc[:, (comp2, slice(None))].abs().mean().mean() combined_channel = {} for i, chan1 in enumerate(channel_mean.index): for chan2 in channel_mean.index[i+1:]: combined_channel[f"{chan1} + {chan2}"] = shap_df.loc[:, (slice(None), chan1)].abs().mean().mean() + \ shap_df.loc[:, (slice(None), chan2)].abs().mean().mean() all_compartment_importance = list(compartment_mean.values) + list(combined_compartment.values()) all_compartment_labels = list(compartment_mean.index) + list(combined_compartment.keys()) all_channel_importance = list(channel_mean.values) + list(combined_channel.values()) all_channel_labels = list(channel_mean.index) + list(combined_channel.keys()) create_extended_radar_plot(all_compartment_importance, all_compartment_labels, "SHAP Importance by Compartment (Individual and Combined)") create_extended_radar_plot(all_channel_importance, all_channel_labels, "SHAP Importance by Channel (Individual and Combined)") return merged_df
interperate_vision_model = interpret_vision_model