"""Find screen phenotypes and test which guides or genes explain them.
WHAT IT IS FOR
==============
The **Regression** tile opens this module because
:func:`perform_regression` is its main entry point: it joins per-well image
scores to sequencing counts and estimates guide- or gene-level associations
for a pooled screen. The module also contains a separate classical
machine-learning workflow, :func:`generate_ml_scores`, which trains a
classifier on measured single-object features and turns those predictions
into the score table a regression can consume.
WHAT IT NEEDS
=============
A regression run is configured with ``paired_data``: ordered score/count CSV
pairs whose plate and well identities agree. Older ``score_data`` and
``count_data`` lists are migrated positionally, but explicit pairs are safer.
Choose the score column with ``dependent_variable`` and provide the relevant
control wells, plate metadata, analysis ``level`` (guide, gene, or both),
multiple-testing threshold, and either a supported ``regression_type`` or
``None`` for distribution-based selection. Family-specific settings are
validated rather than silently ignored. The classical-ML path instead needs
one or more ``measurements.db`` files, labelled positive and negative controls
or an annotation column, and a model choice such as XGBoost, logistic
regression, or random forest.
WHAT IT PRODUCES
================
Regression results go into a new, non-overwriting
``<output root>/results/<analysis kind>[_n]`` directory. ``results.csv`` is
the primary combined table; ``results_grna.csv`` and ``results_gene.csv``
make the fitted levels explicit, and ``results_significant.csv`` records
thresholded hits. The same directory holds summaries, diagnostics, volcano
and plate figures, optional publication-panel packages, resource measurements,
or a detailed failure report. A requested level with no fitted rows is kept
as a header-only CSV so downstream tools can distinguish "tested, no rows"
from a missing artifact. Classical ML writes predictions, feature-importance
tables, evaluation results, and a plate heatmap beneath ``results``.
WHAT TO DO NEXT
===============
Read the diagnostic and failure/resource records before ranking hits, then
review the significant table alongside the complete level tables and check
whether guide and gene effects agree. Open the generated result panels for
visual QC and retain the settings/manifests with any reported hit list. If
scores do not yet exist, run :func:`generate_ml_scores`; if counts do not yet
exist, create them with
:func:`spacr.sequencing.generate_barecode_mapping`.
Several statistical distinctions are deliberate. Guide and gene fits are
separate multiple-testing families and receive separate corrections; the
nominal ``alpha``, effect-size threshold, and corrected significance cutoff
are not interchangeable. A mixed model reports guide effects as shrunken
BLUP predictions without guide p- or q-values, so they must not be read as a
second guide significance test. Diagnostic or report-generation failures do
not erase successful scientific output, while an actual regression failure
is recorded and then re-raised unchanged so callers cannot mistake it for a
completed run.
"""
import functools
import logging
import os, sys, re
import pandas as pd
import numpy as np
from scipy import stats
from scipy.stats import shapiro
from math import pi
from sklearn.linear_model import (Lasso, Ridge, LassoCV, RidgeCV,
ElasticNet, ElasticNetCV)
from sklearn.svm import LinearSVC
from sklearn.base import clone
from sklearn.metrics import mean_squared_error
import matplotlib.pyplot as plt
try:
from IPython.display import display
except Exception:
[docs]
def display(*args, **kwargs):
"""Accept and discard display arguments when IPython is unavailable."""
pass
import scipy.stats as st
import statsmodels.api as sm
import statsmodels.formula.api as smf
from statsmodels.regression.mixed_linear_model import MixedLM
from statsmodels.stats.outliers_influence import variance_inflation_factor
from statsmodels.genmod.families import Binomial
from statsmodels.genmod.families.links import Logit
from statsmodels.othermod.betareg import BetaModel
from sklearn.preprocessing import FunctionTransformer
from patsy import dmatrices
from .regression_spec import (DEFAULT_REGRESSION_BACKEND, # noqa: F401
NO_P_VALUE_TYPES,
REGRESSION_BACKENDS,
REGRESSION_BACKEND_ORDER,
REGRESSION_SETTINGS_USED,
REGRESSION_TYPES,
RUN_LEVEL_SETTINGS,
UNSUPPORTED_REGRESSION_TYPES,
_MODEL_LEVEL_DEFAULTS,
_RUN_LEVEL_DEFAULTS)
from .regression_families import (REGRESSION_FAMILY_ASSUMPTIONS, # noqa: F401
REGRESSION_FAMILY_GROUPS,
family_group,
family_label,
regression_family_choices)
from .mixed_gpu import MixedBackendUnavailable # noqa: F401
from .regression_backends import (backend_label, # noqa: F401
backend_status,
backend_supports,
resolve_backend_name)
from sklearn.model_selection import StratifiedKFold
from sklearn.feature_selection import SelectKBest, f_classif
from sklearn.ensemble import RandomForestClassifier, HistGradientBoostingClassifier
from sklearn.linear_model import LogisticRegression
from sklearn.inspection import permutation_importance
from sklearn.metrics import classification_report, precision_recall_curve
from sklearn.preprocessing import StandardScaler
from sklearn.preprocessing import MinMaxScaler
from scipy.spatial.distance import cosine, euclidean, mahalanobis, cityblock, minkowski, chebyshev, braycurtis
from xgboost import XGBClassifier
from . import frame_handoff, schema, tabular
from .openmp_guard import single_threaded_openmp, guarded_n_jobs
from .plot import save_figure
LOG = logging.getLogger("spacr.ml")
_FLOWVIEW_TRUE_VALUES = frozenset({"1", "on", "true", "yes"})
def _flowview_event(action, *args):
"""Reach optional Classify tracing without importing it when disabled."""
trace_module = sys.modules.get("spacr.flowview.trace")
if trace_module is None:
enabled_by_environment = os.environ.get("SPACR_FLOWVIEW", "")
if enabled_by_environment.strip().casefold() not in _FLOWVIEW_TRUE_VALUES:
return False
try:
from .flowview import trace as trace_module
except BaseException:
return False
try:
if not trace_module.is_enabled():
return False
from .flowview import _classify_stages
return bool(getattr(_classify_stages, f"_{action}")(*args))
except BaseException:
return False
def _flowview_pipeline(family):
"""Finish or fail the active graph without changing scientific output."""
def decorate(function):
"""Return a metadata-preserving lifecycle wrapper for ``function``."""
@functools.wraps(function)
def observed(*args, **kwargs):
"""Call unchanged, reporting success or failure to an active trace."""
settings = args[0] if args else kwargs.get("settings")
active = _flowview_event("begin", settings, family)
try:
result = function(*args, **kwargs)
except BaseException as scientific_error:
if active:
_flowview_event("fail", scientific_error)
raise
if active:
_flowview_event("finish")
return result
return observed
return decorate
def _flowview_advance(node_id):
"""Record one real operation boundary, or do nothing when disabled."""
_flowview_event("advance", node_id)
def _flowview_metric(name, value):
"""Record one scalar on the active stage, or do nothing when disabled."""
_flowview_event("metric", name, value)
from scipy.stats import kstest, normaltest
import matplotlib
from .figures.style import ROLES, figure_style, theme_target
if not (sys.platform.startswith(('win', 'darwin')) or os.environ.get('DISPLAY')):
matplotlib.use('Agg')
import warnings
def _require_backend(regression_type, regression_backend):
"""Resolve and validate a regression backend for a model family.
Validate at run time because settings loaded from a file can bypass the
GUI's disabled backend entries. Never substitute another backend silently.
:param regression_type: the family being fitted.
:param regression_backend: name or label, or ``None`` for the default.
:returns: the canonical backend name.
:raises ValueError: when the backend cannot fit the family, is not
installed, or needs a GPU that is not there.
"""
name = resolve_backend_name(regression_backend)
if name == DEFAULT_REGRESSION_BACKEND:
return name
status = backend_status(name, regression_type)
if not status['enabled']:
raise ValueError(
f"{status['reason']} Set regression_backend='statsmodels' to fit "
f"it with the default backend, which produced every existing "
f"result.")
return name
def _say_what_a_mixed_fit_will_cost(backend, df=None):
"""Describe the expected cost of a statsmodels mixed fit before it starts.
Print nothing for non-default backends. For statsmodels, include the row
count when available and state whether the compatible Torch GPU backend
can be selected instead.
"""
if backend != DEFAULT_REGRESSION_BACKEND:
return
rows = None
try:
rows = len(df) if df is not None else None
except TypeError: # noqa: BLE001
rows = None
try:
status = backend_status('torch', 'mixed')
available = bool(status.get('enabled'))
reason = str(status.get('reason') or '')
except Exception: # noqa: BLE001
available, reason = False, ''
size = f" on {rows} wells" if rows else ""
print(f"Fitting the mixed model{size} with statsmodels. This is the slow "
f"one: dense linear algebra, measured at 54x OLS on 40 genes and "
f"rising with screen size, and it prints nothing while it runs.")
if available:
print(" The same model, same estimates, is available on the GPU: set "
"regression_backend='torch'. Measured on this screen: 26 "
"seconds against >25 minutes.")
elif reason:
print(f" The GPU backend would be faster but is not usable here: "
f"{reason}")
#: Settings that must lie strictly inside 0 and 1, and what each one is for.
#: Checked BEFORE the run writes anything -- see
#: :func:`_reject_impossible_probabilities`.
_UNIT_INTERVAL_SETTINGS = {
'fdr_alpha': "the family-level rejection threshold for adjusted P values",
'p_threshold_alpha': "the cut applied to the P value column",
'alpha': None,
}
def _reject_impossible_probabilities(settings):
"""Validate probability thresholds before the run writes output.
Check every setting in :data:`_UNIT_INTERVAL_SETTINGS` that represents a
probability and reject non-numeric values or values outside the open
interval ``(0, 1)``. The penalty parameter named ``alpha`` is deliberately
excluded because it is not a probability.
"""
for key, what in _UNIT_INTERVAL_SETTINGS.items():
if what is None or key not in settings:
continue
value = settings.get(key)
if value is None:
continue
try:
number = float(value)
except (TypeError, ValueError):
raise ValueError(
f"{key}={value!r} is not a number. It is {what}, and must "
f"be strictly between 0 and 1 (usually 0.05).") from None
if not (0.0 < number < 1.0):
raise ValueError(
f"{key}={number!r} is outside 0 and 1. It is {what}, so it "
f"has no meaning there; the usual value is 0.05. "
+ ("A value one less than what you meant is what a spin box "
"does when its arrow or the scroll wheel is nudged -- "
f"{number + 1:g} may be the number you set."
if -1.0 < number < 0.0 else
"Set it in the Significance section, or in "
"settings/regression.csv if this run came from a file."))
def _concat_named_csvs(paths):
"""Read every CSV in ``paths`` into one frame.
A screen's counts and scores are one file per plate, and this question
is asked of the screen rather than of a plate, so they are read together.
Each file goes through :func:`spacr.tabular.read_table`, so a header
spelled ``column_name`` or ``Well`` reaches the fit under the canonical
key names.
:param paths: one path, a list of them, or nothing.
:returns: the concatenated frame.
:raises ValueError: when there is nothing readable to concatenate.
"""
import pandas as pd
from .tabular import read_table
if not paths:
raise ValueError("no table was given to read")
if isinstance(paths, (str, os.PathLike)):
paths = [paths]
frames = []
for one in paths:
try:
frames.append(read_table(one))
except Exception as exc:
raise ValueError(f"{one} could not be read: {exc}") from exc
if not frames:
raise ValueError("no table was given to read")
return pd.concat(frames, ignore_index=True)
def _well_block_tokens(settings, key):
"""The row/column/well tokens one control-block setting names.
:param settings: the run's settings mapping.
:param key: ``'positive_control_wells'`` or its negative twin.
:returns: the tokens, lower-cased, with the empty ones dropped.
A list or a bare string, because both spellings reach here: the panel
writes a list and a settings CSV can carry either.
"""
raw = (settings or {}).get(key)
if raw is None:
return []
values = raw if isinstance(raw, (list, tuple, set)) else [raw]
return [str(value).strip().lower() for value in values
if str(value).strip()]
def _wells_in_block(labels, tokens):
"""Which of ``labels`` the block ``tokens`` name.
:param labels: well labels as the score table carries them, ``prc`` form
-- ``plate_row_column``.
:param tokens: what the plate design calls the block: a column (``c2``),
a row (``r1``) or a whole well (``plate1_r1_c2``).
:returns: the matching labels, sorted and de-duplicated.
THE TOKEN IS MATCHED AGAINST A PART, NOT AS A SUBSTRING. ``'c2' in
'plate1_r1_c20'`` is true and says nothing, so a plate wider than nine
columns would fold column 20 into column 2's reference and shift the
endpoint the whole calibration is anchored on.
"""
wanted = set(tokens)
out = set()
for label in labels:
text = str(label).strip()
parts = {part.strip().lower() for part in text.split('_')}
parts.add(text.lower())
if parts & wanted:
out.add(text)
return sorted(out)
def _calibration_inputs(settings):
"""Gather what the fraction-threshold sweep needs, from the run's own files.
THE IMAGING SIDE IS THE CLASSIFIER SCORE. `mixed_ratio_calibration` takes
an ``(n_cells, n_features)`` block and one well label per cell, and asks
what mixture of positive and negative control each well looks like. The
per-cell score column is exactly that measurement with one feature, so
the score table the run already loaded is the imaging side and no second
source is needed.
THE PURE WELLS ARE NAMED FROM THE PLATE DESIGN, through
`positive_control_wells` and `negative_control_wells`. Identifying them by
their reported fraction would be circular: that fraction is the quantity
under test, and a bias large enough to matter pushes a pure well the
wrong side of any cut-off.
THE WELLS AND THE GUIDE ARE TWO SETTINGS, and reading one for the other is
what stopped this running at all. `positive_control_id` is a gene or gRNA ID
SUBSTRING in a regression -- it defaults to '239740' -- and was being
matched against well labels, which no well label has ever contained. So
every screen that ticked the box was refused with "no well matched", and
the three control-block settings that exist to answer this were never
read. The guide is what `positive_guide` needs; the wells are what
`pure_pc_wells` needs.
:param settings: the regression settings.
:returns: keyword arguments for
:func:`spacr.fraction_calibration.sweep_fraction_threshold`.
:raises ValueError: when the screen cannot answer the question -- no
control-well block named, no positive-control guide, or no score
column to read. The caller turns that into a printed reason and the
threshold the settings gave.
"""
import numpy as np
positive_wells = _well_block_tokens(settings, 'positive_control_wells')
negative_wells = _well_block_tokens(settings, 'negative_control_wells')
if not positive_wells or not negative_wells:
raise ValueError(
"the plate design names no positive_control_wells and "
"negative_control_wells, and a control-well calibration has "
"nothing to calibrate against")
positive_guide = str(settings.get('positive_control_id') or '').strip()
if not positive_guide:
raise ValueError(
"positive_control_id names no gRNA, so there is no guide whose "
"sequenced share can be compared with the imaging")
counts = _concat_named_csvs(settings.get('count_data'))
scores = _concat_named_csvs(settings.get('score_data'))
well_column = str(settings.get('count_well_column') or 'prc')
score_column = str(settings.get('dependent_variable') or 'pred')
if score_column not in scores.columns:
raise ValueError(
f"the score table has no {score_column!r} column to read the "
f"imaging side from")
if well_column not in scores.columns:
raise ValueError(
f"the score table has no {well_column!r} column, so a cell "
f"cannot be placed in a well")
usable = scores[[well_column, score_column]].dropna()
features = np.asarray(usable[score_column], dtype=float).reshape(-1, 1)
wells = [str(w) for w in usable[well_column]]
pure_pc = _wells_in_block(wells, positive_wells)
pure_nc = _wells_in_block(wells, negative_wells)
if not pure_pc or not pure_nc:
raise ValueError(
f"no well matched {positive_wells} and {negative_wells}, so "
f"there is no pure control to anchor the fit")
return {
"counts": counts,
"features": features,
"wells": wells,
"positive_guide": positive_guide,
"pure_pc_wells": pure_pc,
"pure_nc_wells": pure_nc,
"normalise": bool(settings.get('normalise_fraction', True)),
"well_column": well_column,
"guide_column": str(settings.get('count_grna_column') or 'grna'),
"count_column": str(settings.get('count_value_column') or 'count'),
}
def _calibrated_fraction_threshold(settings):
"""The cut-off the control wells imply, or ``None`` if they cannot say.
Returns ``None`` -- rather than raising -- for every reason the sweep
might not apply: the plate design names no pure control wells, there
are too few of them to fit anything, the counts are missing the columns
it reads, or the optional module is not importable. Each of those is an
ordinary answer to "can this screen calibrate itself", and none of them
is a reason to stop a run that already had a usable threshold.
:param settings: the regression settings, read for the control-well
names and the count table.
:returns: the measured threshold, or ``None``.
"""
try:
from .fraction_calibration import sweep_fraction_threshold
except Exception:
print("fraction-threshold calibration is unavailable; "
"using the threshold as given")
return None
try:
result = sweep_fraction_threshold(**_calibration_inputs(settings))
except (KeyError, ValueError, TypeError) as exc:
print(f"fraction-threshold calibration did not run: {exc}")
return None
chosen = result.get("chosen") if isinstance(result, dict) else None
if chosen is None:
print("fraction-threshold calibration found no cut-off it preferred; "
"using the threshold as given")
return None
try:
from .fraction_calibration import describe
print(describe(result))
except Exception:
print(f"fraction_threshold calibrated to {chosen}")
return float(chosen)
def _graph_sequencing_stats(settings):
"""Resolve the sequencing threshold helper through one testable seam."""
from .sequencing import graph_sequencing_stats
return graph_sequencing_stats(settings)
#: File types the run treats as a figure when it collects what a helper drew.
_FIGURE_SUFFIXES = ('.pdf', '.png', '.svg', '.jpg', '.jpeg', '.tif', '.tiff',
'.eps')
def _screen_figure_folders(settings):
"""Where a sequencing helper drops its figures: beside the COUNT DATA.
`graph_sequencing_stats` writes ``<count folder>/results/`` for the
threshold sweep and ``<count folder>/`` for the unique-count plate
heatmap, both derived from ``settings['count_data'][0]`` inside
:mod:`spacr.sequencing`. Neither is the run's own folder, which is the
whole problem this list exists to solve.
"""
folders = []
for path in (settings.get('count_data') or []):
base = os.path.dirname(str(path))
for candidate in (base, os.path.join(base, 'results')):
if candidate and candidate not in folders:
folders.append(candidate)
return folders
def _figure_stamps(folders):
"""``{path: (mtime, size)}`` for every figure directly inside ``folders``.
Not recursive, and not a bare listing: a run of the same screen writes
the same file NAMES, so identity has to include the stamp or a figure
left by yesterday's run reads as one this run drew.
"""
stamps = {}
for folder in folders:
try:
entries = list(os.scandir(folder))
except OSError:
continue
for entry in entries:
if not entry.name.lower().endswith(_FIGURE_SUFFIXES):
continue
try:
if not entry.is_file():
continue
info = entry.stat()
except OSError:
continue
stamps[entry.path] = (info.st_mtime_ns, info.st_size)
return stamps
def _keep_figures_with_the_run(before, folders, destination):
"""Copy newly written figures into the run-specific output folder.
Compare current file stamps with ``before`` and copy only new or changed
figures. Retain the originals because other workflows may reference the
screen-level folder.
:param before: the stamps from :func:`_figure_stamps` taken first.
:param folders: the same folders it was taken over.
:param destination: the run folder.
:returns: the paths written, so the caller can name them.
"""
import shutil
kept = []
for path, stamp in sorted(_figure_stamps(folders).items()):
if before.get(path) == stamp:
continue
target = os.path.join(destination, os.path.basename(path))
if os.path.abspath(target) == os.path.abspath(path):
continue
try:
os.makedirs(destination, exist_ok=True)
shutil.copy2(path, target)
except OSError as error:
print(f"Could not keep {os.path.basename(path)} with the run: "
f"{error}")
continue
kept.append(target)
return kept
def _run_random_state(default=None):
"""Return the active run's seed, for an estimator's ``random_state=``.
Imported inside the call rather than at module scope: :mod:`spacr.runctx`
reaches :mod:`spacr.settings`, which reaches back here, and a top-level
import would be a cycle. Outside a run this is whatever ``default`` was,
which is the literal these call sites used to hard-code.
:param default: the value to use when no run is open.
:returns: the run seed, or ``default``.
"""
from .runctx import random_state
return random_state(default)
warnings.filterwarnings("ignore", message="3D stack used, but stitch_threshold=0 and do_3D=False, so masks are made per plane only")
class _DispersedVariance:
"""Scale a statsmodels variance function by a constant dispersion factor.
``Binomial.__init__`` stores a ``varfuncs`` callable in the *instance*
``__dict__`` under the name ``variance``, and an instance attribute
always wins over a subclass method of the same name. Overriding
``variance`` in a subclass therefore has no effect on anything
statsmodels does. Wrapping the stored callable is the only way to make
the factor reach the fit, and delegating attribute lookups keeps
``family.variance.deriv`` — which ``GLM`` calls — working.
:param varfunc: The variance callable installed by statsmodels.
:param dispersion: Multiplicative variance scaling.
"""
def __init__(self, varfunc, dispersion):
"""Store the variance callable and its multiplicative dispersion."""
self._varfunc = varfunc
self.dispersion = dispersion
def __call__(self, mu):
"""Return ``dispersion * varfunc(mu)``."""
return self.dispersion * self._varfunc(mu)
def deriv(self, mu):
"""Return the dispersion-scaled derivative of the variance function."""
return self.dispersion * self._varfunc.deriv(mu)
def __getattr__(self, name):
"""Delegate every other attribute to the wrapped variance function.
Raises ``AttributeError`` - never ``KeyError`` - when ``_varfunc`` is
not set yet, so ``copy``/``pickle`` can probe for ``__setstate__`` and
friends on a half-built instance without blowing up.
"""
try:
varfunc = self.__dict__['_varfunc']
except KeyError:
raise AttributeError(name) from None
return getattr(varfunc, name)
[docs]
class QuasiBinomial(Binomial):
"""Binomial GLM family scaled by a dispersion parameter (quasi-binomial).
:param link: statsmodels link instance. Default ``Logit()``.
:param dispersion: Multiplicative variance scaling. Default ``1.0``.
"""
def __init__(self, link=Logit(), dispersion=1.0):
"""Store the dispersion factor after delegating to ``Binomial``."""
super().__init__(link=link)
self.dispersion = dispersion
[docs]
self.variance = _DispersedVariance(self.__dict__['variance'], dispersion)
def variance(self, mu):
"""Adjust the variance with the dispersion parameter.
:param mu: fitted mean probabilities, scalar or array; the binomial
variance of ``mu`` is multiplied by ``dispersion``.
"""
return self.dispersion * super().variance(mu)
[docs]
def calculate_p_values(X, y, model):
"""Return OLS-style p-values for a fitted model's coefficients.
**These are not valid frequentist p-values for a penalised fit**, and the
two callers that reach them know it in different ways. The standard error
is the unpenalised ``rse * sqrt(diag((X'X)^-1))`` while the coefficient it
is divided into has been shrunk, so the test is mis-specified. The
direction of the error is the one that matters here and it is the safe one:
the penalty shrinks the numerator and inflates the residual in the
denominator, so the statistic is too SMALL and the p-value too large. A
penalised fit under-detects here; it does not manufacture hits.
``lasso`` and ``elasticnet`` do not rely on this at all —
:data:`NO_P_VALUE_TYPES` routes them to a bootstrap selection frequency
instead. ``ridge`` does, because it never sets a coefficient to exactly
zero and so has no selection frequency to report (every feature would score
1.0), and a conservative test is a better answer than no test.
``tests/test_regression_orientation.py`` pins the null case, which is where
an anticonservative version of this would show.
:param X: Design matrix (``n x p``).
:param y: Observed responses.
:param model: Fitted estimator exposing ``predict`` and ``coef_``.
:returns: 1D array of length ``p``; entries are ``NaN`` when
``n <= p + 1``.
"""
y_true = np.asarray(y).ravel()
y_pred = np.asarray(model.predict(X)).ravel()
residuals = y_true - y_pred
dof = X.shape[0] - X.shape[1] - 1
if dof <= 0:
return np.full(X.shape[1], np.nan)
residual_std_error = np.sqrt(np.sum(residuals ** 2) / dof)
XtX = X.T @ X
try:
XtX_inv = np.linalg.inv(np.asarray(XtX))
except np.linalg.LinAlgError:
XtX_inv = np.linalg.pinv(np.asarray(XtX))
se = residual_std_error * np.sqrt(np.diag(XtX_inv))
coefs = np.asarray(model.coef_).ravel()
with np.errstate(divide='ignore', invalid='ignore'):
t_stats = np.where(se > 0, coefs / se, 0.0)
p_values = 2 * (1 - st.norm.cdf(np.abs(t_stats)))
return p_values
def _is_out_of_memory(exc) -> bool:
"""Is this exception a device or host memory exhaustion?
Matched by NAME as well as by type, because `torch.cuda.OutOfMemoryError`
only exists once torch is imported and this must not import it to find
out. A plain `MemoryError` counts too -- `mixed_gpu` raises one
deliberately when the design will not fit.
"""
if isinstance(exc, MemoryError):
return True
name = type(exc).__name__
if "OutOfMemory" in name:
return True
text = str(exc).lower()
return "out of memory" in text or "cuda error: out of memory" in text
[docs]
def create_volcano_filename(csv_path, regression_type, alpha, dst):
"""Build the path this run's volcano plot will be saved to.
Path construction only: nothing is read, written or created, and the
``.pdf`` in the name is not binding - :func:`spacr.plot.save_figure`
rewrites the extension to whichever format the figure preference selected.
:param csv_path: Source CSV. Only its basename with the last extension
stripped becomes the ``<name>_volcano_plot.pdf`` stem, and only its
directory is used, when ``dst`` is falsy. The file is never opened, so
a path that does not exist is fine; a bare filename yields a bare
relative result rather than a path under the working directory.
:param regression_type: Prefixed to the filename, unless it is exactly
``'quantile'`` - then ``alpha`` is prefixed instead. ``None`` is
stamped literally, giving ``None_...``: :func:`regression` calls this
before :func:`check_distribution` resolves the auto-selected model, so
an auto run's plot is never named for the model it actually fitted.
:param alpha: Read only on the ``'quantile'`` branch; accepted and ignored
for every other type, whatever its value. :func:`regression` passes
the ``quantile`` setting here, not the penalty, so two quantiles of
one screen cannot overwrite each other.
:param dst: Output directory. Any falsy value, ``None`` and ``''`` alike,
falls back to the directory of ``csv_path``. It is not created here.
:returns: The joined path, which :func:`regression` hands to
:func:`spacr.plot.volcano_plot` as ``save_path``.
"""
volcano_filename = os.path.splitext(os.path.basename(csv_path))[0] + '_volcano_plot.pdf'
volcano_filename = f"{regression_type}_{volcano_filename}" if regression_type != 'quantile' else f"{alpha}_{volcano_filename}"
if dst:
return os.path.join(dst, volcano_filename)
return os.path.join(os.path.dirname(csv_path), volcano_filename)
[docs]
def scale_variables(X, y):
"""Min-max scale the independent (X) and dependent (y) variables to [0, 1].
Constant columns are passed through UNCHANGED. ``MinMaxScaler`` maps a
column with zero range to all-zeros, and patsy's intercept is exactly such
a column, so scaling a design matrix used to silently delete its
intercept: statsmodels then fitted a model through the origin and still
printed an ``Intercept`` row, of 0.000, in the summary. Every coefficient
in that fit absorbs the mean it can no longer estimate.
:param X: Design matrix (DataFrame).
:param y: Response, as a 2-D array or single-column frame.
:returns: ``(X_scaled, y_scaled)`` - a DataFrame with ``X``'s columns and
a 2-D ``numpy`` array.
Example:
.. code-block:: python
X = pd.DataFrame({'Intercept': 1.0, 'a': [1.0, 2.0, 3.0]})
scale_variables(X, np.array([[0.0], [1.0], [2.0]]))[0]['Intercept']
# -> 1.0, 1.0, 1.0 (not 0.0, 0.0, 0.0)
"""
scaler_X = MinMaxScaler()
scaler_y = MinMaxScaler()
X_scaled = pd.DataFrame(scaler_X.fit_transform(X), columns=X.columns)
constant = X.nunique(dropna=False) <= 1
for column in X.columns[constant.values]:
X_scaled[column] = np.asarray(X[column], dtype=float)
y_scaled = scaler_y.fit_transform(y)
return X_scaled, y_scaled
[docs]
def select_glm_family(y):
"""Choose a ``statsmodels`` GLM family from the range and type of the response.
A coarser rule than :func:`pick_glm_family_and_link`, which also sets
the link: binary values give ``Binomial``, any other values inside
``[0, 1]`` give ``QuasiBinomial``, non-negative integers give
``Poisson`` and everything else ``Gaussian``.
:param y: Response vector.
:returns: An unfitted ``statsmodels`` family instance on its default link.
"""
if np.all((y == 0) | (y == 1)):
print("Using Binomial family (for binary data).")
return sm.families.Binomial()
elif (y >= 0).all() and (y <= 1).all():
print("Using Quasi-Binomial family (for proportion data including 0 and 1).")
return QuasiBinomial()
elif np.all(y.astype(int) == y) and (y >= 0).all():
print("Using Poisson family (for count data).")
return sm.families.Poisson()
else:
print("Using Gaussian family (for continuous data).")
return sm.families.Gaussian()
#: The two things a fixed-effects screen model can be ABOUT, and the term each
#: one regresses on. One level per fit -- see :func:`prepare_formula`.
LEVEL_TERMS: dict = {
'grna': 'fraction:grna',
'gene': 'gene_fraction:gene',
}
#: What ``level`` may be. ``'both'`` is not a design; it is an instruction to
#: fit BOTH of the above SEPARATELY and correct each within itself.
LEVEL_CHOICES: tuple = ('both', 'grna', 'gene')
#: Deprecated formula fragment that combines guide and gene fractions.
#:
#: ``check_and_clean_data`` builds ``gene_fraction`` as the SUM of the gene's
#: gRNA fractions within a well, so every ``gene_fraction:gene[G]`` column is
#: the sum of gene G's ``fraction:grna`` columns whenever G's guides do not
#: share a well. Combining both terms therefore creates exact linear
#: dependencies and a non-identifiable design. The literal remains available
#: so spaCR can detect and refuse that formula explicitly.
COLLINEAR_FORMULA_FRAGMENT = 'fraction:grna + gene_fraction:gene'
def _level_term(level):
"""``LEVEL_TERMS[level]``, with the error that says why ``'both'`` is not one.
:raises ValueError: for ``'both'`` (two fits, so ask for one at a time) or
for anything that is not a level at all.
"""
key = str(level).strip().lower()
if key in LEVEL_TERMS:
return LEVEL_TERMS[key]
if key == 'both':
raise ValueError(
"level='both' runs two fits, not one design, so it has no single "
"formula: call prepare_formula once with level='grna' and once "
"with level='gene'. Putting both terms in one design is the "
"collinear model: gene_fraction is the sum of the gene's gRNA "
"fractions, so the gene block is an exact linear combination of "
"the gRNA block and the coefficients are not identifiable.")
raise ValueError(
f"level={level!r} is not a model level. Choose one of "
f"{LEVEL_CHOICES!r}.")
#: What the intercept of a screen regression may be asked to be.
#:
#: 'fitted' the model estimates it, which is what every fit did before
#: this was a choice;
#: 'zero' no intercept at all -- the fit passes through the origin, so
#: a guide's coefficient is its whole predicted score rather
#: than a departure from a baseline;
#: 'control' the response is centred on the negative controls before
#: fitting, so the intercept IS the control level and every
#: coefficient reads as "above or below the controls";
#: 'value' the number the user gives. The response is shifted by it and
#: the term is suppressed, which pins the intercept at exactly
#: that value rather than estimating one near it.
INTERCEPT_MODES = ("fitted", "zero", "control", "value")
[docs]
def centre_on_controls(df, dependent_variable, nc):
"""Subtract the negative controls' median response. Returns (df, offset).
THIS IS WHAT MAKES THE INTERCEPT MEAN SOMETHING. A fitted intercept is
the response where every predictor is zero, which on a screen design is
a well with no guide in it -- a point that does not exist. Centred on
the negative controls, the intercept is the control level, and every
coefficient reads directly as "this far above or below the controls".
The offset is returned rather than swallowed so the caller can report
it: a coefficient table whose response was shifted, with nothing saying
by how much, is a table nobody can compare with another run.
:param df: the long frame the fit runs on.
:param dependent_variable: the response column.
:param nc: the negative-control guide or gene, as the settings name it.
:returns: ``(frame, offset)``. The frame is a copy when it was changed
and the original when it was not; ``offset`` is 0.0 when no control
row could be identified, and the caller is expected to say so.
"""
import numpy as _np
if not nc or dependent_variable not in getattr(df, "columns", ()):
return df, 0.0
wanted = str(nc).strip().lower()
if not wanted:
return df, 0.0
mask = None
for column in ("grna", "gene", "grna_name", "gene_name"):
if column not in df.columns:
continue
found = df[column].astype(str).str.strip().str.lower() == wanted
mask = found if mask is None else (mask | found)
if mask is None or not bool(mask.any()):
return df, 0.0
values = _np.asarray(df.loc[mask, dependent_variable], dtype=float)
values = values[_np.isfinite(values)]
if not values.size:
return df, 0.0
offset = float(_np.median(values))
if offset == 0.0:
return df, 0.0
shifted = df.copy()
shifted[dependent_variable] = (
_np.asarray(shifted[dependent_variable], dtype=float) - offset)
return shifted, offset
[docs]
def screen_is_blockable(df) -> bool:
"""Whether ``screenID`` can be a term in this frame's design.
True only when the column exists and carries more than one distinct
value. A single-screen project is the normal case and must be untouched
by the design: it has no screenID at all, or one value, and either
way the term would be a constant column.
The same rule :func:`spacr.measurement_scan._dummy_block` applies, stated
once for the formula path so a frame cannot be blocked on by one and not
the other.
:param df: the design DataFrame, or ``None`` (returns ``False``); its
``screenID`` column is compared as strings.
"""
from .schema import SCREEN_KEY
if df is None or SCREEN_KEY not in getattr(df, 'columns', ()):
return False
return int(df[SCREEN_KEY].astype(str).nunique(dropna=True)) > 1
#: How a guide's BLUP is named in the coefficient table. NOT ``fraction:grna``
#: -- a BLUP is a shrunken prediction of a random effect, not a fixed
#: coefficient, and giving it the fixed term's name is exactly how it would end
#: up in a hit list with a q value beside it.
BLUP_FEATURE_TEMPLATE = 'blup:grna[{}]'
#: What each row of a mixed fit's coefficient table IS. The column exists so
#: nothing downstream has to guess from the name, and so a variance component
#: or a BLUP can never be read as an effect on the response.
TERM_FIXED = 'fixed'
TERM_VARIANCE = 'variance'
TERM_BLUP = 'random_effect_blup'
def _blup_guide_name(key):
"""The guide id inside a statsmodels variance-component BLUP key.
``vc_formula={'grna': '0 + C(grna)'}`` labels its columns
``grna[C(grna)[244480_3]]``, so the id is the innermost bracket.
Returns ``None`` for the group's own intercept (``'Group'``) and for
anything that is not a guide component.
"""
text = str(key)
match = re.search(r'C\(grna\)\[(?:T\.)?([^\]]+)\]', text)
if match:
return match.group(1)
return None
def _answering_stop(model):
"""Add a cancellation checkpoint to a statsmodels model instance.
Wrap the instance's ``loglike`` method because optimizers evaluate it on
each step, providing finer cancellation granularity than the fit callback.
The wrapper propagates :class:`spacr.cancellation.PipelineCancelled` and
returns the same model instance.
"""
from .cancellation import checkpoint
original = model.loglike
def loglike(*args, **kwargs):
"""Check for cancellation, then return the original likelihood result."""
checkpoint()
return original(*args, **kwargs)
model.loglike = loglike
return model
[docs]
def fit_mixed_model(df, formula, dst, *, random_row_column_effects=False,
gene_column='gene', guide_column='grna',
regression_backend=DEFAULT_REGRESSION_BACKEND):
"""Fit a mixed model with guides nested within genes.
The model treats genes as fixed effects and guides as random effects
nested within genes. In statsmodels notation, ``groups=gene`` supplies the
outer random intercept and ``vc_formula={'grna': '0 + C(grna)'}`` supplies
the guide-within-gene variance component.
A blockable ``screenID`` supplied by :func:`prepare_formula` remains a
fixed effect. With only two screen levels, a random screen variance would
be estimated from one degree of freedom. The plate is not nested within
the screen because plate position is already represented by the row and
column structure. Single-screen data omit the constant screen term to
avoid a rank-deficient design.
Parameters
----------
df : pandas.DataFrame
Model data containing the formula variables and the gene and guide
grouping columns.
formula : str
Fixed-effects formula, normally returned by :func:`prepare_formula`
with ``level='gene'``.
dst : path-like
Destination for the residual histogram.
random_row_column_effects : bool, default False
Add row and column variance components instead of fixed terms.
gene_column : str, default 'gene'
Column containing the outer gene groups.
guide_column : str, default 'grna'
Column containing guides nested within each gene.
regression_backend : {'statsmodels', 'torch'}, default 'statsmodels'
Mixed-model backend. The torch backend fits the same nested model
with GPU acceleration when available.
Returns
-------
mixed_model
Fitted backend-specific mixed-model result.
coef_df : pandas.DataFrame
Fixed effects, variance components, and guide BLUPs. Variance
components and BLUPs have ``NaN`` p-values because they are not
fixed-effect hypothesis tests.
Raises
------
ValueError
If required grouping columns are missing, no gene has multiple
guides, or the backend cannot fit the nested design.
MixedBackendUnavailable
If the selected mixed-model backend is unavailable.
"""
from .plot import plot_histogram
for column in (gene_column, guide_column):
if column not in df.columns:
raise ValueError(
f"the mixed model nests {guide_column!r} inside "
f"{gene_column!r}, and this frame has no {column!r} column. "
f"Columns: {sorted(df.columns)[:20]}")
response = str(formula).split('~', 1)[0].strip() or 'the response'
groups = _mixed_model_groups(df, response, df.index,
gene_column=gene_column)
guides_per_gene = df.groupby(gene_column, observed=True)[
guide_column].nunique()
if int((guides_per_gene > 1).sum()) == 0:
raise ValueError(
f"the mixed model nests guides inside genes, and no gene in this "
f"frame has more than one guide ({len(guides_per_gene)} genes, "
f"one guide each). The guide variance component would be exactly "
f"confounded with the residual and would come back as zero. Use a "
f"fixed-effects regression_type with level='gene' -- with one "
f"guide per gene the two levels are the same model anyway.")
vc_formula = {guide_column: f'0 + C({guide_column})'}
if random_row_column_effects:
vc_formula['rowID'] = '0 + C(rowID)'
vc_formula['columnID'] = '0 + C(columnID)'
backend = _require_backend('mixed', regression_backend)
_say_what_a_mixed_fit_will_cost(backend, df)
try:
if backend == 'torch':
from .mixed_gpu import mixedlm_torch
mixed_model = mixedlm_torch(formula, df, groups,
vc_formula=vc_formula)
print(mixed_model.summary_line())
else:
model = smf.mixedlm(formula, data=df, groups=groups,
re_formula='1', vc_formula=vc_formula)
mixed_model = _answering_stop(model).fit()
except MixedBackendUnavailable:
raise
except Exception as error:
raise ValueError(
f"MixedLM could not fit y ~ gene_fraction:gene + (1 | "
f"{gene_column}/{guide_column}) on this frame: "
f"{type(error).__name__}: {error}. The nesting needs several "
f"genes, several guides inside at least some of them, and more "
f"wells than genes. Choose a fixed-effects regression_type with "
f"level='gene' or level='grna' if this screen cannot support "
f"it.") from error
df['residuals'] = mixed_model.resid
plot_histogram(df, 'residuals', dst=dst)
fixed_names = set(map(str, mixed_model.fe_params.index))
coefs = mixed_model.params
p_values = mixed_model.pvalues
term_types = [TERM_FIXED if str(name) in fixed_names else TERM_VARIANCE
for name in coefs.index]
parameter_p = np.asarray(p_values.values, dtype=float)
parameter_p = np.where(
np.array(term_types) == TERM_VARIANCE, np.nan, parameter_p)
frames = [pd.DataFrame({
'feature': [str(name) for name in coefs.index],
'coefficient': np.asarray(coefs.values, dtype=float),
'p_value': parameter_p,
'term_type': term_types,
})]
blups = {}
for group_key, values in (mixed_model.random_effects or {}).items():
for key, value in dict(values).items():
guide = _blup_guide_name(key)
if guide is None:
continue
blups[guide] = float(value)
if blups:
guides = sorted(blups)
frames.append(pd.DataFrame({
'feature': [BLUP_FEATURE_TEMPLATE.format(g) for g in guides],
'coefficient': [blups[g] for g in guides],
'p_value': np.full(len(guides), np.nan, dtype=float),
'term_type': [TERM_BLUP] * len(guides),
}))
coef_df = pd.concat(frames, ignore_index=True)
n_blups = int((coef_df['term_type'] == TERM_BLUP).sum())
print(f"Mixed model fitted by regression_backend={backend_label(backend)}")
print(f"Mixed model: gene fixed, guide random nested in gene "
f"({groups.nunique()} genes, {n_blups} guide BLUPs). "
f"A BLUP has no p-value, so results_grna.csv from a mixed run is a "
f"shrunken prediction per guide and carries no q value.")
if not bool(getattr(mixed_model, 'converged', True)):
variances = ', '.join(
f"{name}={value:.3g}" for name, value in
zip(vc_formula, np.atleast_1d(np.asarray(mixed_model.vcomp,
dtype=float))))
print("\n"
" ###############################################################\n"
" # WARNING: the mixed model did not converge. #\n"
" ###############################################################\n"
f" Variance components: {variances}; group variance "
f"{float(np.asarray(mixed_model.cov_re).ravel()[0]):.3g}.\n"
" A variance on the boundary at zero is the usual cause and is\n"
" itself an answer, but the standard errors and p-values of the\n"
" gene fixed effects are not trustworthy while it stands. Fit a\n"
" fixed-effects regression_type with level='gene' to get gene\n"
" effects whose intervals can be reported.\n")
return mixed_model, coef_df
[docs]
def check_and_clean_data(df, dependent_variable):
"""Prepare the merged count / score frame for model fitting.
Drops rows with a missing ``fraction`` or dependent variable, casts
the identifier columns to categorical and reports (without dropping)
collinear columns via VIF. The returned frame keeps only
``fraction``, the dependent variable, ``gene``, ``grna``, ``prc``,
``plateID``, ``rowID``, ``columnID``, and ``cell_count`` and ``screenID``
when present, plus a computed ``gene_fraction`` column: the sum of the gene's gRNA
fractions within each well, which the regression formula regresses on.
:param df: Merged DataFrame of counts and scores.
:param dependent_variable: Name of the response column.
:returns: The cleaned DataFrame used as the model input.
:raises ValueError: if a ``(prc, grna)`` pair carries more than one
``fraction``, which makes ``gene_fraction`` ambiguous.
"""
def handle_missing_values(df, columns):
"""Handle missing values in specified columns."""
missing_summary = df[columns].isnull().sum()
print("Missing values summary:")
print(missing_summary)
df_cleaned = df.dropna(subset=columns).copy()
if df_cleaned.shape[0] < df.shape[0]:
print(f"Dropped {df.shape[0] - df_cleaned.shape[0]} rows with missing values in {columns}.")
return df_cleaned
def ensure_valid_types(df, columns):
"""Ensure that specified columns are categorical."""
for col in columns:
if not isinstance(df[col].dtype, pd.CategoricalDtype):
df[col] = pd.Categorical(df[col])
print(f"Converted {col} to categorical type.")
return df
def check_collinearity(df, columns):
"""Check for collinearity using VIF (Variance Inflation Factor)."""
print("Checking for collinearity...")
df_encoded = df[columns]
df_encoded = df_encoded.apply(pd.to_numeric, errors='coerce')
if np.linalg.matrix_rank(df_encoded.values) < df_encoded.shape[1]:
print("Warning: Perfect multicollinearity detected! Dropping correlated columns.")
df_encoded = df_encoded.loc[:, ~df_encoded.columns.duplicated()]
vif_data = pd.DataFrame()
vif_data["Feature"] = df_encoded.columns
try:
vif_data["VIF"] = [variance_inflation_factor(df_encoded.values, i) for i in range(df_encoded.shape[1])]
except np.linalg.LinAlgError:
print("LinAlgError: Unable to compute VIF due to matrix singularity.")
return df_encoded
print("Variance Inflation Factor (VIF) for each feature:")
print(vif_data)
high_vif_columns = vif_data[vif_data["VIF"] > 10]["Feature"].tolist()
if high_vif_columns:
print(f"Warning: high collinearity (VIF > 10) for: {high_vif_columns}. "
f"Keeping them - the regression formula requires both - but "
f"coefficient estimates may be unstable.")
return df_encoded
df = handle_missing_values(df, ['fraction', dependent_variable])
df = ensure_valid_types(df, ['grna', 'gene', 'plateID', 'rowID', 'columnID', 'prc'])
df_cleaned = check_collinearity(df, ['fraction', dependent_variable])
df_cleaned['gene'] = df['gene']
df_cleaned['grna'] = df['grna']
df_cleaned['prc'] = df['prc']
df_cleaned['plateID'] = df['plateID']
df_cleaned['rowID'] = df['rowID']
df_cleaned['columnID'] = df['columnID']
if 'cell_count' in df.columns:
df_cleaned['cell_count'] = df['cell_count']
from .schema import SCREEN_KEY
if SCREEN_KEY in df.columns:
df_cleaned[SCREEN_KEY] = df[SCREEN_KEY]
grna_key = ['prc', 'gene', 'grna']
per_grna = df_cleaned[grna_key + ['fraction']].drop_duplicates()
clash = per_grna.duplicated(subset=grna_key, keep=False)
if clash.any():
offenders = per_grna.loc[clash, grna_key].drop_duplicates()
raise ValueError(
f"{len(offenders)} (well, gRNA) pair(s) carry more than one "
f"'fraction', so the gene's share of the well is ambiguous - e.g. "
f"{offenders.iloc[0].to_dict()}. This means the count table was "
f"joined twice, or two count files describe the same plate. "
f"Aggregate the counts per (prc, grna) before regressing.")
gene_totals = per_grna.groupby(['prc', 'gene'], observed=False)['fraction'].sum()
df_cleaned['gene_fraction'] = pd.MultiIndex.from_arrays(
[df_cleaned['prc'], df_cleaned['gene']]).map(gene_totals)
print("Data is ready for model fitting.")
return df_cleaned
[docs]
def minimum_cell_simulation(settings, num_repeats=10, sample_size=100, tolerance=0.02, smoothing=10, increment=10, dst=None):
"""
Estimate the minimum number of cells per well needed for a stable well mean.
For the wells with the most objects, repeatedly subsamples cells at
increasing sample sizes and records the mean absolute difference from
the well's full mean. Plots the smoothed curve with a ±1 s.d. band,
marks the elbow point (or ``settings['min_cells_per_well']`` when it is
set) and writes ``cell_min_threshold.pdf`` into ``dst``.
Pass ``dst`` to keep the figure in a specific run folder. When omitted,
the function uses the screen-level ``results`` folder derived from
``count_data`` for compatibility with direct notebook and script calls.
:param settings: Requires ``score_data`` (CSV path or list of paths),
``dependent_variable``, ``tolerance`` (int percent or float
fraction) and
``min_cells_per_well``. ``count_data`` is needed only when ``dst`` is
left unset, and only to locate the figure.
:param num_repeats: Subsamples drawn per sample size. Default ``10``.
:param sample_size: Number of wells, taken largest-first by cell
count, to simulate. Default ``100``.
:param tolerance: Unused; the tolerance applied is
``settings['tolerance']``.
:param smoothing: Rolling-window width used to smooth the curve.
:param increment: Step between the simulated sample sizes.
:param dst: Folder for ``cell_min_threshold.pdf``, created if missing.
Default ``None``: ``<folder of settings['count_data'][0]>/results``.
:returns: The elbow point's sample size, i.e. the minimum cell count
per well, for passing to :func:`process_scores`.
:raises ValueError: if ``settings['tolerance']`` is neither an int nor
a float.
"""
from .utils import correct_metadata_column_names
if isinstance(settings['score_data'], str):
settings['score_data'] = [settings['score_data']]
dfs = []
for i, score_data in enumerate(settings['score_data']):
df = tabular.read_table(score_data)
df = correct_metadata_column_names(df)
df['plateID'] = f'plate{i + 1}'
if 'prc' not in df.columns:
df['prc'] = _compose_prc_column(df)
dfs.append(df)
df = pd.concat(dfs, axis=0)
cell_counts = df.groupby('prc').size().reset_index(name='cell_count')
top_wells = cell_counts.nlargest(sample_size, 'cell_count')['prc']
df = df[df['prc'].isin(top_wells)]
diff_data = []
for i, (prc, group) in enumerate(df.groupby('prc')):
original_mean = group[settings['dependent_variable']].mean()
max_cells = len(group)
sample_sizes = np.arange(2, max_cells + 1, increment)
for sample_size in sample_sizes:
abs_diffs = []
for _ in range(num_repeats):
sample = group.sample(n=sample_size, replace=False)
sampled_mean = sample[settings['dependent_variable']].mean()
abs_diff = abs(sampled_mean - original_mean)
abs_diffs.append(abs_diff)
avg_abs_diff = np.mean(abs_diffs)
diff_data.append((sample_size, avg_abs_diff))
diff_df = pd.DataFrame(diff_data, columns=['sample_size', 'avg_abs_diff'])
summary_df = diff_df.groupby('sample_size').agg(
mean_abs_diff=('avg_abs_diff', 'mean'),
std_abs_diff=('avg_abs_diff', 'std')
).reset_index()
summary_df['smoothed_mean_abs_diff'] = summary_df['mean_abs_diff'].rolling(window=smoothing, min_periods=1).mean()
if isinstance(settings['tolerance'], int):
tolerance_fraction = settings['tolerance'] / 100
elif isinstance(settings['tolerance'], float):
tolerance_fraction = settings['tolerance']
else:
raise ValueError("Tolerance must be an integer 0 - 100 or float 0.0 - 1.0.")
relative_thresholds = {
prc: tolerance_fraction * group[settings['dependent_variable']].mean()
for prc, group in df.groupby('prc')
}
summary_df['relative_threshold'] = summary_df['sample_size'].map(
lambda size: np.mean([relative_thresholds[prc] for prc in top_wells])
)
elbow_df = summary_df[summary_df['smoothed_mean_abs_diff'] <= summary_df['relative_threshold']]
if not elbow_df.empty:
elbow_point = elbow_df.iloc[0]
else:
elbow_point = summary_df.iloc[-1]
if dst is None:
dst = os.path.join(os.path.dirname(settings['count_data'][0]),
'results')
dst = os.path.abspath(os.path.expanduser(os.fspath(dst)))
os.makedirs(dst, exist_ok=True)
mark = (elbow_point['sample_size'] if settings['min_cells_per_well'] is None
else settings['min_cells_per_well'])
fig_file_path = _draw_the_cell_count_sweep(
summary_df, mark, os.path.join(dst, 'cell_min_threshold.pdf'))
if fig_file_path:
print(f"Saved {fig_file_path}")
return elbow_point['sample_size']
def _statsmodels_p_values(model, coefs):
"""Return per-coefficient p-values from a statsmodels-shaped results object.
Every statsmodels results class spaCR fits exposes ``pvalues``. The
fallback exists for :mod:`spacr.power_model`, whose Laplace approximation
reports standard errors rather than a test: a two-sided normal p-value
from ``coef / bse`` is exactly what a Wald test on that approximation is,
and computing it here keeps the horseshoe fit in the same table as the
rest instead of giving it a private code path.
:param model: Fitted results object.
:param coefs: Its ``params``, already extracted.
:returns: 1-D float array aligned with ``coefs``.
:raises ValueError: when the object carries neither ``pvalues`` nor
``bse``, so no inference is possible.
"""
pvalues = getattr(model, 'pvalues', None)
if pvalues is not None:
return np.asarray(pvalues, dtype=float).reshape(-1)
bse = getattr(model, 'bse', None)
if bse is None:
raise ValueError(
f"{type(model).__name__} exposes neither .pvalues nor .bse, so "
f"spaCR cannot attach a p-value to its coefficients. A results "
f"object handed to process_model_coefficients must carry one or "
f"the other.")
std_err = np.asarray(bse, dtype=float).reshape(-1)
with np.errstate(divide='ignore', invalid='ignore'):
z = np.where(std_err > 0,
np.asarray(coefs, dtype=float).reshape(-1) / std_err,
0.0)
return 2.0 * (1.0 - st.norm.cdf(np.abs(z)))
def _bootstrap_wald_p_values(model, X, y, n_boot=200, random_state=0):
"""Return bootstrap Wald p-values for an estimator with no inference.
Refits ``model``'s estimator on ``n_boot`` nonparametric resamples of the
rows, takes the empirical standard deviation of each coefficient across
the resamples and reports ``2 * (1 - Phi(|coef| / sd))``.
This is the honest minimum for the hinge backend: an SVM has no
likelihood, so there is no Wald or likelihood-ratio test to run, and the
alternative - leaving ``p_value`` NaN - would make
:func:`perform_regression` select ``p_value <= 0.05`` on an all-NaN column
and report "0 significant gRNAs" for every hinge run, which reads exactly
like a screen with no hits.
A resample that loses a class entirely is skipped rather than fitted; a
coefficient whose bootstrap standard deviation is zero (never selected, or
identical in every resample) gets ``p = 1``, never a division by zero.
:param model: A fitted scikit-learn estimator; cloned, never refitted in
place, so the caller's model object is untouched.
:param X: Design matrix.
:param y: Response the model was fitted on - for hinge, the BINARISED one.
:param n_boot: Number of resamples. Default 200.
:param random_state: Seed, so a hit list is reproducible from the settings.
:returns: 1-D float array of length ``X.shape[1]``.
:raises RuntimeError: when no resample could be fitted at all.
"""
rng = np.random.default_rng(random_state)
X_values = np.asarray(X, dtype=float)
y_values = np.asarray(y, dtype=float).reshape(-1)
n = X_values.shape[0]
draws = []
one_class = 0
unfittable = 0
last_failure = None
for _ in range(int(n_boot)):
idx = rng.integers(0, n, size=n)
y_boot = y_values[idx]
if np.unique(y_boot).size < 2:
one_class += 1
continue
try:
fitted = clone(model).fit(X_values[idx], y_boot)
except Exception as exc:
unfittable += 1
last_failure = exc
continue
draws.append(np.asarray(fitted.coef_, dtype=float).ravel())
if not draws:
raise RuntimeError(
f"none of the {n_boot} bootstrap resamples could be fitted, so no "
f"standard error is available for the hinge coefficients. This "
f"usually means one class holds only a handful of wells; check "
f"hinge_threshold.")
dropped = int(n_boot) - len(draws)
if dropped:
LOG.warning(
"hinge bootstrap: %d of %d resamples produced no coefficients "
"(%d were one-class, %d would not fit%s). The p-values below are "
"computed from the remaining %d.",
dropped, int(n_boot), one_class, unfittable,
f"; last error: {last_failure}" if last_failure is not None else "",
len(draws))
if len(draws) < 2:
LOG.warning(
"hinge bootstrap: only %d resample(s) survived, so the coefficient "
"standard deviation is zero and EVERY p-value below is exactly "
"1.0. That is an absence of evidence, not evidence of absence — "
"do not read it as 'no significant gRNAs'.", len(draws))
coefs = np.asarray(model.coef_, dtype=float).ravel()
sd = np.std(np.vstack(draws), axis=0, ddof=1) if len(draws) > 1 else \
np.zeros_like(coefs)
with np.errstate(divide='ignore', invalid='ignore'):
z = np.where(sd > 0, coefs / sd, 0.0)
return 2.0 * (1.0 - st.norm.cdf(np.abs(z)))
#: Backends whose fitted results object carries ``params`` and ``pvalues``
#: directly. Most are statsmodels; ``horseshoe`` and ``rra`` are spaCR's own
#: adapters (:class:`_HorseshoeResults`, :class:`_RRAResults`), which exist so
#: that a model with a posterior or a permutation null lands in the same table
#: as the likelihood fits instead of getting a private code path.
#: ``mixed`` is here too, and its variance components are dropped below - they
#: are not effects on the response.
_STATSMODELS_COEF_TYPES = (
'ols', 'wls', 'rlm', 'huber', 'glm', 'poisson', 'logit', 'probit',
'quasi_binomial', 'quantile', 'mixed', 'horseshoe', 'rra',
'spline',
)
#: Backends whose fitted object exposes ``coef_`` and ``predict`` and carries
#: no inference of its own, so :func:`calculate_p_values` supplies the
#: (deliberately conservative) p-value. Three are scikit-learn's; ``group_lasso``
#: is :class:`_GroupLassoResults` around :mod:`spacr.group_lasso`, and it is
#: here rather than in a branch of its own precisely so it reports what the
#: other penalised backends report.
_SKLEARN_COEF_TYPES = ('ridge', 'lasso', 'elasticnet', 'group_lasso')
#: The level term patsy writes for one gRNA or one gene:
#: ``fraction:grna[224750_2]`` or ``gene_fraction:gene[T.224750]``. Anchored,
#: so a nuisance column -- ``Intercept``, ``rowID[T.r2]``, ``columnID[T.c7]``,
#: ``screenID[T.b]`` -- does not match and is answered with None. An unanchored
#: search would read ``r2`` out of ``rowID[T.r2]`` and hand the row and column
#: dummies to the grouping as though they were genes.
_LEVEL_TERM_IN_FEATURE = re.compile(
r'^(?:fraction:grna|gene_fraction:gene)\[(?:T\.)?(.*)\]$')
#: The guide number a gRNA id ends with: ``224750_2`` -> gene ``224750``.
_GUIDE_SUFFIX = re.compile(r'_\d+$')
def _gene_of_design_column(column):
"""The gene a design column belongs to, or ``None`` for a nuisance term.
THE GROUPING BOTH NEW BACKENDS RUN ON. ``group_lasso`` penalises a gene's
guide columns as one block and ``rra`` aggregates their ranks, so both need
to know which columns are the same gene's -- and the design matrix is all
either of them is given. It is parsed from the column name rather than
passed in, because the name is what patsy actually built the column from;
a second, separately supplied grouping could disagree with it and would
then split a gene silently.
``perform_regression`` reduces ``TGGT1_224750_2`` to ``224750_2`` before
the fit (the three-token org/gene/guide split it makes on the merged
frame), but a caller that fits ``regression_model`` directly may not have,
so the trailing guide number is stripped from whatever is there and the
ORG PREFIX IS LEFT ALONE: it is constant across a screen, so ``TGGT1_224750`` and ``224750``
are each a consistent key for their own frame, and stripping it would be a
guess about naming.
:param column: a design-matrix column name.
:returns: the gene id, or ``None`` when the column is not a level term.
"""
text = str(column)
match = _LEVEL_TERM_IN_FEATURE.match(text)
if match is None:
return None
identifier = match.group(1)
if text.startswith('gene_fraction:'):
return identifier
return _GUIDE_SUFFIX.sub('', identifier) or identifier
def _level_term_mask(columns):
"""A boolean mask of the columns that name a gRNA or a gene.
``dtype=bool`` is not incidental: an empty design gives ``np.array([])``,
which is float64, and boolean-indexing a coefficient vector with a float
array raises ``IndexError`` instead of selecting nothing.
"""
return np.array([_gene_of_design_column(column) is not None
for column in columns], dtype=bool)
def _design_column_groups(columns):
"""One group label per design column, nuisance terms in groups of their own.
:mod:`spacr.group_lasso` groups by label, so every column needs one. A
nuisance column gets its OWN name as its label, which makes it a singleton
group: it is penalised on its own, exactly as ordinary lasso would, and it
can never be pulled into a gene's block and dragged to zero with it.
"""
return [_gene_of_design_column(column) or str(column)
for column in columns]
def _say_when_a_control_matched_nothing(coef_df, nc, pc, controls) -> None:
"""Warn, by name, about a control that selected no coefficient.
The consequence is named because the number is not obviously missing:
the run completes, the volcano draws, and the effect-size cut is simply
measured on nothing.
"""
counts = coef_df['condition'].value_counts()
for value, tag, what in ((nc, 'nc', 'negative_control_id'),
(pc, 'pc', 'positive_control_id')):
if value in (None, '') or int(counts.get(tag, 0)):
continue
print(f" WARNING: {what}={value!r} matches no coefficient in this "
f"screen, so the baseline and the effect-size cut that read it "
f"have nothing to measure. Check it against the guide names in "
f"the count table -- spaCR reads a bare id as a GENE and one "
f"with an underscore as a GUIDE.")
if (controls or []) and not int(counts.get('control', 0)):
print(f" WARNING: none of the {len(list(controls))} control(s) named "
f"matches a coefficient, so there is no control spread to "
f"measure an effect-size cut on.")
[docs]
def label_control_condition(features, guides, nc=None, pc=None, controls=None,
*, strict: bool = False, verbose: bool = False):
"""Label every coefficient row ``'nc'``, ``'pc'``, ``'control'`` or ``'other'``.
The ``condition`` column: what the volcano colours by, what the results
panel offers in "colour by", and -- the reason this is a function rather
than four lines inside :func:`process_model_coefficients` -- what the
EFFECT-SIZE CUT measures its spread on. A coefficient table without it is
a table :meth:`spacr.qt.widgets.regression_results.RegressionResultsPanel.
set_threshold_method` answers "No control coefficients, so no effect-size
cut" for, which is what every guide-permutation run used to get.
Precedence is ``nc``, then ``pc``, then the explicit ``controls`` list, so
a guide named in two of them is reported once and always the same way.
:param features: the model term per row, e.g. ``fraction:grna[000000_1]``.
``nc`` and ``pc`` are matched as SUBSTRINGS of it, which is how a
negative control given as a gene id reaches a term named for a guide.
:param guides: the guide identifier per row, matched whole against
``controls``. Both sides are compared as text, so a control list that
round-tripped through a settings CSV as integers still matches.
:param nc: negative-control identifier, or ``None`` for no negative
control.
:param pc: positive-control identifier, or ``None``.
:param controls: non-targeting guide identifiers, or ``None``. ``None``
means "no control list" and labels nothing -- it is the value
:func:`perform_regression` documents for a control-free screen, and
the inline version this replaced raised ``TypeError`` on it.
:param strict: raise :class:`spacr.control_names.ControlNotFound` when a
NAMED ``nc`` or ``pc`` matches nothing. Off by default so a call on a
partial frame is not an error; the run turns it on, because there a
control matching nothing is a number computed against an empty set.
:param verbose: print what each control resolved to and how much it
matched.
:returns: a :class:`pandas.Series` of labels aligned with ``features``.
"""
features = pd.Series(features).astype(str)
guides = pd.Series(guides).astype(str)
guides.index = features.index
nc_name = '' if nc is None else str(nc)
pc_name = '' if pc is None else str(pc)
control_names = {str(name) for name in (controls or [])}
from .control_names import rows_for
labels = pd.Series('other', index=features.index, dtype=object)
genes = guides.fillna('').str.split('_').str[0]
library = list(guides.astype(str).unique())
if control_names:
for name in sorted(control_names):
mask, note = rows_for(name, guides, genes, names=library)
if verbose and note:
print(f" {note}")
labels[mask.to_numpy()] = 'control'
for name, tag in ((pc_name, 'pc'), (nc_name, 'nc')):
if not name:
continue
mask, note = rows_for(name, guides, genes, names=library,
strict=strict,
label='positive control' if tag == 'pc'
else 'negative control')
if verbose and note:
print(f" {note}")
labels[mask.to_numpy()] = tag
return labels
[docs]
def process_model_coefficients(model, regression_type, X, y, nc, pc, controls,
hinge_threshold=None, hinge_n_boot=200):
"""Return a DataFrame of model coefficients and p-values, one row per term.
Every name in :data:`REGRESSION_TYPES` has a branch here. It is the same
table for all of them - ``feature``, ``coefficient``, ``p_value``,
``-log10(p_value)``, ``grna``, ``condition`` - because everything
downstream (the volcano plot, the hit table, the metadata merge) reads
those columns and nothing else.
:param model: The fitted object from :func:`regression_model`.
:param regression_type: Which backend produced it.
:param X: Design matrix, used for the sklearn feature names and for the
p-value approximations that need the data back.
:param y: Response, likewise.
:param nc: Negative-control identifier, matched against the feature name.
:param pc: Positive-control identifier.
:param controls: Explicit list of control gRNA identifiers.
:param hinge_threshold: The binarisation cut used by the hinge fit; the
bootstrap below must reproduce the SAME two classes the fit saw.
:param hinge_n_boot: Bootstrap resamples used for the hinge p-values.
:returns: Coefficient DataFrame with the row/column nuisance terms removed.
:raises ValueError: on an unsupported ``regression_type``.
"""
if regression_type == 'beta':
coefs = model.params
std_err = model.bse
wald_stats = coefs / std_err
p_values = 2 * (1 - st.norm.cdf(np.abs(wald_stats)))
coef_df = pd.DataFrame({
'feature': coefs.index,
'coefficient': coefs.values,
'std_err': std_err.values,
'wald_stat': wald_stats.values,
'p_value': p_values,
})
elif regression_type in _STATSMODELS_COEF_TYPES:
coefs = model.params
p_values = _statsmodels_p_values(model, coefs)
coef_df = pd.DataFrame({
'feature': coefs.index,
'coefficient': coefs.values,
'p_value': np.asarray(p_values, dtype=float),
})
if regression_type == 'mixed':
coef_df = coef_df[coef_df['feature'].isin(
[str(c) for c in X.columns])].reset_index(drop=True)
elif regression_type in _SKLEARN_COEF_TYPES:
coefs = np.asarray(model.coef_).ravel()
p_values = calculate_p_values(X, y, model)
coef_df = pd.DataFrame({
'feature': X.columns,
'coefficient': coefs,
'p_value': p_values,
})
elif regression_type == 'hinge':
coefs = np.asarray(model.coef_).ravel()
p_values = _bootstrap_wald_p_values(
model, X, binarise_response(y, hinge_threshold,
name='dependent variable'),
n_boot=hinge_n_boot)
coef_df = pd.DataFrame({
'feature': X.columns,
'coefficient': coefs,
'p_value': p_values,
})
else:
raise ValueError(f"Unsupported regression type: {regression_type}")
coef_df['-log10(p_value)'] = -np.log10(coef_df['p_value'])
coef_df['grna'] = (
coef_df['feature']
.str.extract(r'\[(.*?)\]')[0]
.str.replace(r'^T\.', '', regex=True)
)
coef_df['condition'] = label_control_condition(
coef_df['feature'], coef_df['grna'], nc=nc, pc=pc, controls=controls,
verbose=True)
_say_when_a_control_matched_nothing(coef_df, nc, pc, controls)
nuisance = coef_df['feature'].astype(str).str.match(
r'^(?:plateID|rowID|columnID|screenID)\[')
return coef_df[~nuisance]
def _draw_the_threshold_sweep(settings, res_folder, *,
measured: bool = False) -> None:
"""Draw the guide-fraction sweep without replacing the threshold in force.
:param settings: the regression settings, read for the threshold in force
and the count tables the sweep reads.
:param res_folder: this run's folder, kept in the signature because the
caller collects what the sweep drew into it.
:param measured: the threshold in force came from the control-well
calibration rather than from the user.
THE TWO ANSWERS, SIDE BY SIDE, AND NAMED. The sweep answers "how many
guides per well do I want" from the counts alone; the calibration answers
"which cut-off makes imaging and sequencing agree" from the control
wells. They are different questions, so neither replaces the other and
the run reports both -- but a run that says "you set 0.0168" about a
number the calibration measured is telling the user something untrue
about where their threshold came from, which is the one thing this line
exists to say.
Plotting is diagnostic; a rendering failure is reported without
invalidating the regression run.
"""
try:
chosen = settings.get('fraction_threshold')
derived = _graph_sequencing_stats(settings)
if derived is not None and chosen is not None:
whose, mine = (("the control-well calibration measured",
"The measured value") if measured
else ("you set", "Your value"))
print(f"gRNA fraction-threshold sweep drawn: {whose} "
f"{chosen}; the sweep's own pick on this screen is "
f"{derived}. The two answer different questions -- which "
f"cut-off the control wells agree at, and how many guides "
f"a well should keep -- so neither replaces the other. "
f"{mine} is the one in force.")
except Exception as error: # noqa: BLE001
print(f"the gRNA fraction-threshold sweep could not be drawn "
f"({type(error).__name__}: {error}); the run is unaffected "
f"and fraction_threshold={settings.get('fraction_threshold')} "
f"is still in force")
def _show_response_distribution(before_df, dependent_variable, settings):
"""Display the response distribution before and after transformation.
The panel is emitted through Matplotlib so the Qt bridge can add it to the
figure queue. It is also shown when no transformation is selected, making
the unchanged distribution explicit.
"""
if before_df is None or not settings.get('plot', True):
return
try:
import matplotlib.pyplot as plt
from .response_distribution import panel
wanted = [str(dependent_variable),
str(settings.get('dependent_variable') or "")]
column = next((c for c in wanted if c and c in before_df), None)
if column is None:
numeric = [c for c in before_df.columns
if pd.api.types.is_numeric_dtype(before_df[c])]
column = numeric[-1] if numeric else None
if column is None:
print("the response distribution panel was not drawn: the "
"aggregated table carries no numeric response column")
return
_draw_response_panel_in_pyqtgraph(
before_df[column].to_numpy(dtype=float),
str(settings.get('transform') or 'none'), str(column),
settings.get('src'))
except Exception as error: # noqa: BLE001
print(f"the response distribution panel could not be drawn "
f"({type(error).__name__}: {error}); the run is unaffected")
[docs]
def check_distribution(y, epsilon=1e-6):
"""Check the distribution of ``y`` and recommend a regression type.
:param y: Response vector.
:param epsilon: How close to 0 or 1 a value may sit before it counts
as a boundary case. Default ``1e-6``.
:returns: One of ``'logit'``, ``'quasi_binomial'``, ``'beta'``,
``'ols'`` or ``'glm'``, as accepted by :func:`regression`'s
``regression_type``.
"""
if np.all((y == 0) | (y == 1)):
print("Detected binary data.")
return 'logit'
elif (y > 0).all() and (y < 1).all():
if np.any((y < epsilon) | (y > 1 - epsilon)):
print("Detected continuous data near 0 or 1. Using quasi-binomial.")
return 'quasi_binomial'
else:
print("Detected continuous data between 0 and 1 (no boundary issues). Using beta regression.")
return 'beta'
elif (y >= 0).all() and (y <= 1).all():
print("Detected continuous data with boundary values (0 or 1). Using quasi-binomial.")
return 'quasi_binomial'
stat, p_value = stats.normaltest(y)
print(f"Normality test p-value: {p_value:.4f}")
if p_value > 0.05:
print("Detected normally distributed data. Using OLS.")
return 'ols'
if stats.kstest(y, 'beta', args=(2, 2)).pvalue > 0.05:
if np.any((y < epsilon) | (y > 1 - epsilon)):
print("Detected continuous data near 0 or 1. Using quasi-binomial.")
return 'quasi_binomial'
else:
print("Detected continuous data between 0 and 1 (no boundary issues). Using beta regression.")
return 'beta'
print("Detected non-normally distributed data. Using GLM.")
return 'glm'
MIN_POISSON_SAMPLES = 8
def _validate_poisson_response(y, X=None, minimum_samples=MIN_POISSON_SAMPLES,
model="Poisson regression"):
"""Validate a response before fitting a Poisson GLM.
Poisson endog must contain finite, non-negative integer counts. At least
eight observations and one residual degree of freedom are required so
family detection and coefficient inference are not performed on an
undersized or saturated design.
:param y: One-dimensional count response.
:param X: Optional design matrix used to determine the parameter count.
:param minimum_samples: Absolute observation floor.
:param model: What to call the model in the refusal. `horseshoe` is a
sparse Poisson GLM and reaches this validator too, so a user who
chose it was told "Poisson regression requires integer count data" --
an error naming a model they did not ask for, followed by advice
("use a continuous response model") for a choice they never made.
:returns: The validated response as a one-dimensional float array.
:raises ValueError: If the response or sample size is invalid.
"""
try:
counts = np.asarray(y, dtype=float).reshape(-1)
except (TypeError, ValueError) as exc:
raise ValueError(
f"{model} requires numeric count data."
) from exc
if not np.isfinite(counts).all():
raise ValueError(
f"{model} requires finite count data; remove or impute "
"NaN and infinite response values before fitting."
)
if np.any(counts < 0):
raise ValueError(
f"{model} requires non-negative count data; negative "
"response values are not valid counts."
)
if not np.all(np.isclose(counts, np.rint(counts), rtol=0, atol=1e-8)):
raise ValueError(
f"{model} requires integer count data; use a continuous "
"response model for fractional values."
)
if not np.any(counts > 0):
raise ValueError(
f"{model} requires at least one positive count; an "
"all-zero response cannot estimate effects."
)
n_parameters = 0
if X is not None:
x_shape = np.shape(X)
if not x_shape or x_shape[0] != counts.size:
raise ValueError(
f"{model} requires X and y to contain the same "
f"number of observations; got {x_shape[0] if x_shape else 0} "
f"and {counts.size}."
)
n_parameters = 1 if len(x_shape) == 1 else int(x_shape[1])
required = max(int(minimum_samples), n_parameters + 1)
if counts.size < required:
raise ValueError(
f"{model} has too few observations: "
f"received {counts.size}, but at least {required} are required "
f"for {n_parameters} model parameters."
)
return counts
#: Transforms that are THEMSELVES a link function. Applying one and then
#: handing the result to a family whose link does the same job transforms the
#: response twice, and the model fits a quantity nothing measures.
LINK_LIKE_TRANSFORMS = ('log', 'logit')
#: A link-like transform and a GLM's own link are the same operation asked
#: for twice, and there is one right answer: fit the response AS MEASURED
#: and let the family's link do the transforming, once.
#:
#: This used to be a setting with three values. The other two were not
#: choices worth offering: 'transformed' fitted a Gaussian identity model of
#: the transformed response, which is an ordinary linear model and exactly
#: what regression_type='ols' already gives; and 'warn' kept the double
#: transform so an older result could be reproduced. A user who wants the
#: first still has it under its own name, and the second was a bug being
#: preserved.
[docs]
def pick_glm_family_and_link(y, name="", transform=""):
"""Select the GLM family and link that suit the response.
Used by ``regression_type='glm'`` to choose a family from the data rather
than from the user.
:param y: Response vector.
:param name: Response-column name printed with the selected family. The
name makes clear whether the family was chosen from a derived scale.
:param transform: the transform already applied, printed with the name and
checked for the double transform of :func:`double_transform_warning`.
:returns: A ``statsmodels`` family instance with its link set.
:raises ValueError: only through :func:`_validate_poisson_response`, when
the response looks like counts but cannot be one.
"""
family = _choose_glm_family(y, name=name, transform=transform)
warning = double_transform_warning(name, transform, family)
if warning:
print(warning)
return family
def _choose_glm_family(y, name="", transform=""):
"""The family and link, and the sentence saying which scale was examined."""
values = np.asarray(y, dtype=float).reshape(-1)
scale = str(name or 'the response')
if transform:
scale = f"{scale} (after transform={str(transform)!r})"
if np.all((values == 0) | (values == 1)):
print(f"{scale} is binary. Using Binomial family with Logit link.")
return sm.families.Binomial(link=sm.families.links.Logit())
elif (values > 0).all() and (values < 1).all():
print(f"{scale} is strictly between 0 and 1. Using Binomial family "
f"with Logit link; consider regression_type='beta', which models "
f"the variance of a bounded response directly, or "
f"'quasi_binomial' if the wells are overdispersed.")
return sm.families.Binomial(link=sm.families.links.Logit())
elif (values >= 0).all() and (values <= 1).all():
print(f"{scale} is between 0 and 1 including the boundaries. "
f"Using Quasi-Binomial.")
return sm.families.Binomial(link=sm.families.links.Logit())
if (values >= 0).all() and np.all(values.astype(int) == values):
_validate_poisson_response(values, minimum_samples=1)
print(f"{scale} looks like counts. Using Poisson with Log link.")
return sm.families.Poisson(link=sm.families.links.Log())
stat, p_value = normaltest(values)
print(f"Normality test p-value: {p_value:.4f}")
if p_value > 0.05:
print(f"{scale} is normally distributed. Using Gaussian with "
f"Identity link.")
return sm.families.Gaussian(link=sm.families.links.Identity())
if ((values > 0).all()
and kstest(values, 'invgauss', args=(1,)).pvalue > 0.05):
print(f"{scale} looks inverse Gaussian. Using InverseGaussian "
f"with Log link.")
return sm.families.InverseGaussian(link=sm.families.links.Log())
if (values >= 0).all():
print(f"{scale} looks like overdispersed counts. Using Negative "
f"Binomial with Log link.")
return sm.families.NegativeBinomial(link=sm.families.links.Log())
print(f"{scale}: no family fitted the shape, so Gaussian with an "
f"Identity link is used.")
return sm.families.Gaussian(link=sm.families.links.Identity())
[docs]
def binarise_response(y, threshold=None, name='response'):
"""Return ``y`` as a 0/1 vector for a classifier backend, refusing to guess.
The hinge backend fits a decision boundary, so it needs two classes. There
are exactly two ways to get them and this function will not invent a
third:
* ``y`` already holds exactly two distinct finite values (the usual case:
a per-object class call aggregated to a well, or a 0/1 score). The lower
value becomes 0 and the higher becomes 1, so the sign of every
coefficient answers "does this gRNA push wells towards the HIGHER
class", which is the same direction the continuous models report.
* ``threshold`` is given explicitly, and ``y > threshold`` becomes 1.
A continuous response with no threshold is REFUSED. Picking a cut for the
user — the mean, the median, 0.5 — would silently redefine the hypothesis
being tested: on a screen whose well scores run 0.2-0.8 a median split
calls half the plate positive by construction, and the resulting hit list
is a plausible, unfalsifiable artefact of the split.
:param y: Response vector (array, Series or single-column frame).
:param threshold: Explicit cut; values strictly greater become 1.
:param name: Name used in error messages, for a legible failure.
:returns: ``numpy`` float array of 0.0/1.0, same length as ``y``.
:raises ValueError: if ``y`` is continuous and no ``threshold`` is given,
if a given ``threshold`` puts every observation in one class, or if
``y`` holds fewer than two distinct values.
Example:
.. code-block:: python
binarise_response([0, 1, 1, 0]) # -> [0., 1., 1., 0.]
binarise_response([2, 5, 5], ) # -> [0., 1., 1.]
binarise_response([0.2, 0.6], threshold=0.4) # -> [0., 1.]
"""
values = np.asarray(y, dtype=float).reshape(-1)
if not np.isfinite(values).all():
raise ValueError(
f"hinge regression requires a finite {name}; remove or impute the "
f"NaN/infinite values before fitting.")
if _left_blank(threshold):
threshold = None
if threshold is not None:
cut = float(threshold)
binary = (values > cut).astype(float)
n_positive = int(binary.sum())
if n_positive == 0 or n_positive == binary.size:
raise ValueError(
f"hinge_threshold={cut!r} puts all {binary.size} observations "
f"in one class ({name} range "
f"{values.min():.6g}-{values.max():.6g}); a one-class response "
f"has no decision boundary to fit.")
return binary
unique = np.unique(values)
if unique.size == 2:
return (values == unique[1]).astype(float)
if unique.size < 2:
raise ValueError(
f"hinge regression needs two classes but {name} holds the single "
f"value {unique[0]!r}.")
raise ValueError(
f"hinge regression needs a binary {name}, but it holds "
f"{unique.size} distinct values in "
f"{values.min():.6g}-{values.max():.6g}. Set hinge_threshold to the "
f"cut you mean (values strictly above it are the positive class), or "
f"choose a model for a continuous response ('ols', 'beta', "
f"'quantile'). spaCR will not pick the cut for you: a split chosen by "
f"the software decides the hypothesis, not the biology.")
def _left_blank(value) -> bool:
"""Whether a policed setting was left empty rather than answered.
None is the usual empty; ``''`` is what a Qt line edit and a saved
settings CSV produce for the same untouched box; whitespace is what a
hand-edited CSV produces. None of the policed settings takes a string
value, so a blank one can only mean "not answered".
AND NaN, which is the FOURTH spelling of empty and the one that
actually reaches this function. `pandas.read_csv` turns an empty cell
into `float('nan')`, so a settings CSV with `hinge_threshold,` on a line
-- which is what a saved file looks like for every box the user did not
fill -- arrived here as a float. It is not None and not a str, so it
read as "answered", and an ordinary OLS run was refused with
regression_type='ols' does not read hinge_threshold=nan
about a value nobody typed. NaN is never a threshold, a covariance type
or a quantile, so there is no reading of it that means "answered".
"""
if value is None:
return True
if isinstance(value, str):
return not value.strip()
try:
return bool(value != value)
except Exception: # noqa: BLE001
return False
def _reject_unused_settings(regression_type, supplied):
"""Raise when a setting the chosen backend cannot read was set anyway.
``supplied`` maps a setting name to ``(value, default)``. A value equal to
its default is "not asked for" and passes; anything else must appear in
:data:`REGRESSION_SETTINGS_USED` for this type.
Comparing against the default is what makes this usable from a GUI, which
posts every widget on the panel whether or not the user touched it.
A BLANK IS NOT A REQUEST. An empty box in the panel, and the empty cell
a saved settings CSV writes for it, both arrive here as ``''`` -- which
is not equal to a default of ``None`` and was therefore refused. The
symptom was that the screen's OWN saved settings could not be reloaded
and refitted under a different regression type: `hinge_threshold` had
never been typed into, and switching to 'ols' raised on it.
:param regression_type: The backend about to be fitted.
:param supplied: ``{name: (value, default)}`` for the policed settings.
:raises ValueError: naming the setting, the type and the alternative.
"""
used = REGRESSION_SETTINGS_USED.get(regression_type, ())
for name, (value, default) in supplied.items():
if name in used or value == default or _left_blank(value):
continue
raise ValueError(
f"regression_type={regression_type!r} does not read {name}="
f"{value!r}: {_SETTING_NOT_APPLICABLE[name]} Leave {name} at its "
f"default ({default!r}), or choose a regression type that uses it "
f"({', '.join(t for t in REGRESSION_TYPES if name in REGRESSION_SETTINGS_USED[t]) or 'none'}).")
#: Why each policed setting does nothing for the types that do not list it.
#: Split out of :func:`_reject_unused_settings` so the message names the
#: actual reason instead of "not supported".
_SETTING_NOT_APPLICABLE = {
'alpha': "it is the penalty weight of a penalised fit, and this model is "
"unpenalised, so the number would change nothing.",
'l1_ratio': "it splits a penalty between L1 and L2, and only 'elasticnet' "
"has both.",
'cov_type': "it selects a sandwich covariance estimator on a likelihood "
"fit; sklearn's penalised estimators and the robust/quantile "
"fits do not expose one, so the standard errors would come "
"from somewhere other than the label suggests.",
'quantile': "it is the quantile of the conditional distribution being "
"fitted, which only 'quantile' regression has; every other "
"model fits the mean (or the median, for 'rlm').",
'hinge_threshold': "it is the cut that turns a continuous response into "
"the two classes a hinge loss separates; no other "
"model classifies.",
'spline_knots': "it sets how many knots each CONTINUOUS covariate's "
"basis gets, and only the spline fit builds one; the "
"guide columns are untouched either way.",
'spline_degree': "it sets the polynomial degree of that basis, and only "
"the spline fit builds one.",
'huber_t': "it is the residual, in units of the estimated scale, at which "
"Huber's loss switches from squared to linear; only the robust "
"fits have that switch.",
'lasso_n_boot': "it sizes the bootstrap that ranks features by SELECTION "
"frequency, and only a penalty that sets coefficients to "
"exactly zero selects anything - ridge keeps every "
"feature, so its selection frequency is 1.0 by "
"construction, and the likelihood fits are ranked by their "
"own p-values.",
'lasso_selection_threshold': "it is the cut on that same selection "
"frequency, which only the sparse penalties "
"produce.",
'hinge_n_boot': "it sizes the bootstrap that stands in for the standard "
"errors an SVM does not have; every other model reports "
"its own inference.",
'group_lasso_lambda': "it is the block penalty of the group lasso, "
"measured against THIS design's own "
"group_lasso.max_lambda, so it is not the same "
"quantity as 'alpha' and no other model has a "
"block to penalise.",
'rra_alpha': "it is the top fraction of the guide ranking alpha-RRA "
"aggregates over, and only 'rra' ranks anything; every other "
"model estimates coefficients jointly.",
'rra_permutations': "it sizes the permutation null RRA's P value is read "
"off, and every other model gets its P value from a "
"likelihood, a posterior or a bootstrap.",
}
#: The design factors :mod:`pyfixest` absorbs instead of carrying as columns.
#:
#: `prepare_formula` puts ``rowID`` and ``columnID`` in the model as FIXED
#: effects, so patsy dummy-codes them. They are nuisance terms --
#: `process_model_coefficients` drops every one of them from the coefficient
#: table before anybody reads it -- and a nuisance term that is never reported
#: does not have to be a column. Absorbing it by alternating projections
#: (Frisch-Waugh-Lovell) leaves the coefficients that ARE reported unchanged
#: to the last digit and takes the solve down with the design.
#:
#: ``screenID`` is deliberately excluded because it blocks combined-screen
#: fits on the experiment and may be useful in the coefficient table.
_ABSORBED_FIXED_EFFECTS = ('rowID', 'columnID')
def _absorbed_factor_codes(X, factors=_ABSORBED_FIXED_EFFECTS):
"""Recover the level of each factor patsy dummy-coded, per observation.
patsy writes a k-level factor as k-1 indicator columns against a dropped
reference level, so the reference is the row where every one of them is
zero. Reading the codes back out of the design is what lets an absorbing
backend be handed the SAME matrix statsmodels was, rather than the raw
frame -- there is then no second construction of the design to disagree
with the first.
:param X: the design DataFrame patsy built.
:param factors: term names to look for, each dummy-coded as
``name[T.level]``.
:returns: ``(codes, names, n_absorbed_params)`` -- an ``(n, k)`` uint64
array of level codes, the factors actually found, and how many
parameters of the dense design they account for (the intercept plus
each factor's k-1 indicators, which is what the residual degrees of
freedom must still be charged for). ``codes`` is ``None`` when no
factor is present.
:raises ValueError: when a factor's indicator columns are not 0/1, which
means the column named ``rowID[...]`` is not a dummy and absorbing it
would silently fit a different model.
"""
columns = list(getattr(X, 'columns', []))
blocks, names = [], []
n_params = 1 if 'Intercept' in columns else 0
for factor in factors:
prefix = f'{factor}['
block = [c for c in columns if str(c).startswith(prefix)]
if not block:
continue
values = np.asarray(X[block], dtype=float)
if not np.all(np.isin(values, (0.0, 1.0))):
raise ValueError(
f"the design's {factor!r} columns are not 0/1 indicators, so "
f"they cannot be a dummy-coded factor and absorbing them "
f"would fit a different model. Columns: {block[:4]}.")
if np.any(values.sum(axis=1) > 1):
raise ValueError(
f"a row of the design is in more than one {factor!r} level, "
f"so {factor!r} is not a factor and cannot be absorbed.")
blocks.append(np.where(values.any(axis=1), values.argmax(axis=1) + 1,
0).astype(np.uint64))
names.append(factor)
n_params += len(block)
if not blocks:
return None, [], n_params
return np.column_stack(blocks), names, n_params
class _AbsorbedDesign:
"""The design an absorbed fit was run on, in statsmodels' shape.
:mod:`spacr.regression_qc` recovers a design from ``results.model.exog``
and decides the scale rule from ``type(results.model).__mro__`` (see
``regression_qc._model_kind``), so an absorbed fit that carried neither
would lose the diagnostics tab. It carries the FULL design -- the one with
the dummy columns still in it -- because that is the model that was
fitted; absorption is how it was solved, not what it was.
"""
def __init__(self, endog, exog, exog_names):
"""Store response, full design, and statsmodels-compatible names.
:param endog: the response, flattened to one dimension.
:param exog: the FULL design, dummy columns included. Not the
absorbed one: :mod:`spacr.regression_qc` recovers the design from
here, and absorption is how the fit was solved rather than what
was fitted.
:param exog_names: one name per column of ``exog``, in order.
statsmodels' diagnostics index the design by name, so a list that
is shorter than the design silently mislabels the columns after
the gap rather than raising.
"""
self.endog = np.asarray(endog, dtype=float).reshape(-1)
self.exog = np.asarray(exog, dtype=float)
self.exog_names = list(exog_names)
#: ``kind`` -> the design class :mod:`spacr.regression_qc` resolves BY NAME.
#:
#: ``regression_qc._model_kind`` walks ``type(results.model).__mro__`` looking
#: for a class called ``OLS`` or ``WLS``, because ``sm.OLS(...).fit()`` and
#: ``sm.WLS(...).fit()`` share one results class and only ``results.model``
#: tells them apart. An absorbed least-squares fit obeys exactly the scale
#: rule that name selects -- ``scale`` is RSS / (n - p) over the FULL
#: parameter count, and the weighted version is in the metric of
#: ``sqrt(w) * (y - fitted)`` -- so it answers to it. Built with ``type()``
#: rather than written as two ``class`` statements so ``spacr.ml`` does not
#: grow public names ``OLS`` and ``WLS`` that would read as statsmodels'.
_ABSORBED_DESIGN_CLASSES = {
name: type(name, (_AbsorbedDesign,),
{'__doc__': f"The design of an absorbed {name} fit."})
for name in ('OLS', 'WLS')
}
class _AbsorbedLeastSquaresResults:
"""A least-squares fit solved by absorbing its nuisance factors.
Reports what :func:`process_model_coefficients` and
:mod:`spacr.regression_qc` read off a statsmodels results object --
``params``, ``bse``, ``pvalues``, ``tvalues``, ``resid``,
``fittedvalues``, ``scale``, ``df_resid`` -- for the coefficients that
SURVIVE absorption. The absorbed ones have no row, which is the one way
this fit's answer differs from statsmodels' and is why
``REGRESSION_BACKENDS['pyfixest']['differs']`` says so.
The residuals and ``scale`` are the FULL model's, not the demeaned
regression's: Frisch-Waugh-Lovell makes them the same vector, and the
degrees of freedom are charged for every absorbed parameter, so the
standard errors match statsmodels to the last digit rather than to a
tolerance.
"""
def __init__(self, params, bse, pvalues, resid, fitted, scale,
df_model, df_resid, nobs, model, converged, absorbed,
rsquared):
"""Store absorbed-fit estimates and derive their t statistics.
:param params: coefficients that SURVIVED absorption, indexed by
column name. The absorbed factors have no row here at all, which
is the one way this fit's answer differs from statsmodels'.
:param bse: their standard errors.
:param pvalues: two-sided p values for those coefficients.
:param resid: residuals of the FULL model, not of the demeaned
regression. Frisch-Waugh-Lovell makes them the same vector.
:param fitted: fitted values of the full model.
:param scale: residual variance, charged for every absorbed
parameter, which is what makes the standard errors match
statsmodels to the last digit rather than to a tolerance.
:param df_model: model degrees of freedom, full parameters less the
intercept -- including the absorbed ones.
:param df_resid: residual degrees of freedom after that charge.
:param nobs: number of observations.
:param model: the design object carrying the surviving column names.
:param converged: whether the solver reported convergence.
:param absorbed: the factor names that were absorbed and therefore
have no coefficient. :meth:`predict` refuses a new row by naming
them, because their levels were never estimated.
:param rsquared: R-squared of the full model.
"""
self.params = params
self.bse = bse
self.pvalues = pvalues
with np.errstate(divide='ignore', invalid='ignore'):
self.tvalues = params / bse
self.resid = resid
self.fittedvalues = fitted
self.scale = float(scale)
self.df_model = float(df_model)
self.df_resid = float(df_resid)
self.nobs = float(nobs)
self.model = model
self.converged = bool(converged)
self.absorbed = tuple(absorbed)
self.rsquared = float(rsquared)
def predict(self, exog=None):
"""Fitted values. ``exog`` is accepted and ignored, as sm's OLS does
for the in-sample case; an absorbed fit cannot predict a new row
because it never estimated the absorbed levels."""
if exog is None:
return self.fittedvalues
raise ValueError(
"an absorbed fit did not estimate the levels of "
f"{', '.join(self.absorbed) or 'its nuisance factors'}, so it "
"cannot predict a row it has not seen. Fit with "
"regression_backend='statsmodels' if you need out-of-sample "
"predictions.")
def summary(self):
"""A text summary, so :func:`_write_model_summary` still has one."""
return _AbsorbedSummary(self)
class _AbsorbedSummary:
"""``.as_text()`` for :class:`_AbsorbedLeastSquaresResults`."""
def __init__(self, results):
"""Bind the absorbed-fit results rendered by this summary.
:param results: the :class:`_AbsorbedLeastSquaresResults` to render.
Held, not copied -- the summary is built on demand by
:meth:`_AbsorbedLeastSquaresResults.summary` and read once.
"""
self._results = results
def as_text(self):
"""Return a multiline report of the absorbed least-squares fit."""
r = self._results
lines = [
"Absorbed least squares (pyfixest alternating projections)",
f" observations {int(r.nobs)}",
f" reported coefficients {len(r.params)}",
f" absorbed factors {', '.join(r.absorbed) or 'none'}",
f" residual df {int(r.df_resid)}",
f" error variance {r.scale:.6g}",
f" R-squared {r.rsquared:.6f}",
f" demeaning converged {r.converged}",
"",
"coefficient / std err / t / P>|t|",
]
for name in r.params.index:
lines.append(f" {name} {r.params[name]:.6g} "
f"{r.bse[name]:.6g} {r.tvalues[name]:.4f} "
f"{r.pvalues[name]:.4g}")
return "\n".join(lines)
def __str__(self):
"""Return the same report as :meth:`as_text`."""
return self.as_text()
def _fit_absorbed_least_squares(X, y, weights=None, kind='OLS'):
"""Least squares with ``rowID``/``columnID`` absorbed, via pyfixest.
Use ``pyfixest.core.demean`` to project out row and column factors, then
solve the remaining normal equations by Cholesky decomposition. This
avoids including high-cardinality nuisance dummies in the dense solve
while retaining a coefficient table compatible with the statsmodels path.
:param X: the design DataFrame, dummy columns included.
:param y: the response.
:param weights: per-observation weights for a WLS fit, or ``None``.
:param kind: ``'OLS'`` or ``'WLS'``, which is what
:mod:`spacr.regression_qc` reads to pick its scale rule.
:returns: :class:`_AbsorbedLeastSquaresResults`.
:raises ValueError: when the design carries no absorbable factor (there
is then nothing for this backend to do that statsmodels does not do
better), or when the normal equations are singular.
"""
columns = list(getattr(X, 'columns', []))
if not columns:
raise ValueError(
"the absorbing backend reads which columns are rowID/columnID "
"dummies from the design's COLUMN NAMES, so it needs a DataFrame "
"design; a bare array has no names. Build it with "
"dmatrices(..., return_type='dataframe'), which is what the "
"pipeline hands in.")
codes, absorbed, n_absorbed_params = _absorbed_factor_codes(X)
if codes is None:
raise ValueError(
"regression_backend='pyfixest' absorbs the rowID and columnID "
"fixed effects, and this design has neither -- either "
"model_plate_position=False took them out of the model or the "
"screen sits on one row and one column. There is nothing to "
"absorb, so the fit would be the statsmodels fit with an extra "
"projection in front of it. Set "
"regression_backend='statsmodels'.")
keep = [c for c in columns
if str(c) != 'Intercept'
and not any(str(c).startswith(f'{f}[') for f in absorbed)]
if not keep:
raise ValueError(
"every column of this design is an intercept or an absorbed "
"fixed effect, so the absorbed fit would report no coefficient "
"at all.")
y_flat = np.asarray(y, dtype=float).reshape(-1)
n = y_flat.size
if weights is None:
w = np.ones(n, dtype=float)
else:
w = np.asarray(weights, dtype=float).reshape(-1)
if w.size != n:
raise ValueError(
f"the absorbed fit was given {w.size} weights for {n} "
f"observations.")
if not np.isfinite(w).all() or np.any(w <= 0):
raise ValueError(
"WLS weights must be finite and positive (they are per-well "
f"cell counts); got {np.nanmin(w)}-{np.nanmax(w)}.")
p_full = len(keep) + n_absorbed_params
df_resid = n - p_full
if df_resid <= 0:
raise ValueError(
f"the design has {p_full} parameters ({len(keep)} reported plus "
f"{n_absorbed_params} absorbed) for {n} observations, so there "
f"are no residual degrees of freedom to estimate a standard "
f"error from.")
from pyfixest.core.demean import demean
stacked = np.asfortranarray(
np.column_stack([y_flat, np.asarray(X[keep], dtype=float)]))
demeaned, converged = demean(stacked, codes, w, tol=1e-10)
if not converged:
raise ValueError(
"the alternating projections that absorb "
f"{', '.join(absorbed)} did not converge, so the design was "
"never fully partialled out and the coefficients would not be "
"the least-squares ones. Fit with "
"regression_backend='statsmodels'.")
y_d = demeaned[:, 0]
X_d = demeaned[:, 1:]
Xw = X_d * w[:, None]
xtx = X_d.T @ Xw
xty = Xw.T @ y_d
_rank = int(np.linalg.matrix_rank(xtx))
if _rank < xtx.shape[0]:
raise ValueError(
f"the absorbed design's normal equations are singular "
f"(rank {_rank} of {xtx.shape[0]}), so its {len(keep)} "
f"coefficients are not identified. That is a rank-deficient "
f"design, not a backend failure: statsmodels answers the same "
f"design with a pseudo-inverse, which picks one arbitrary "
f"solution out of infinitely many.")
beta = np.linalg.solve(xtx, xty)
resid = y_d - X_d @ beta
rss = float(resid @ (resid * w))
scale = rss / df_resid
cov = scale * np.linalg.inv(xtx)
se = np.sqrt(np.clip(np.diag(cov), 0.0, None))
with np.errstate(divide='ignore', invalid='ignore'):
t_stats = np.where(se > 0, beta / se, 0.0)
p_values = 2.0 * st.t.sf(np.abs(t_stats), df_resid)
names = [str(c) for c in keep]
params = pd.Series(beta, index=names)
full_resid = resid
fitted = y_flat - full_resid
centred = y_flat - np.average(y_flat, weights=w)
tss = float(centred @ (centred * w))
rsquared = 1.0 - rss / tss if tss > 0 else float('nan')
model = _ABSORBED_DESIGN_CLASSES[kind](
y_flat, np.asarray(X, dtype=float), [str(c) for c in columns])
return _AbsorbedLeastSquaresResults(
params=params,
bse=pd.Series(se, index=names),
pvalues=pd.Series(p_values, index=names),
resid=full_resid, fitted=fitted, scale=scale,
df_model=p_full - 1, df_resid=df_resid, nobs=n, model=model,
converged=converged, absorbed=absorbed, rsquared=rsquared)
#: ``regression_type`` -> the glum family, and how the fit is set up.
#:
#: ``probit`` IS NOT HERE, and that is measured rather than an omission: glum
#: 3.4 ships ``IdentityLink``, ``LogLink``, ``LogitLink``, ``CloglogLink`` and
#: ``TweedieLink`` and has no probit link at all, so a probit fitted "by glum"
#: could only be a logit under the wrong label. ``quasi_binomial`` is not here
#: either -- statsmodels spells it as a Binomial mean with the dispersion
#: taken from the Pearson chi-square (``scale='X2'``), and glum has no
#: equivalent knob, so its standard errors would be the fixed-dispersion ones
#: on a model chosen BECAUSE its dispersion is free. `backend_status` greys
#: the pair out for exactly these reasons.
_GLUM_FAMILIES = {
'poisson': 'poisson',
'logit': 'binomial',
'glm': None,
}
class _GlumResults:
"""A GLM fitted by glum, reporting what statsmodels' GLM results report.
Form covariance from the canonical-link information matrix,
``(X' W X)^-1`` with
``W_ii = v_i (dmu/deta)^2 / V(mu_i)`` and dispersion fixed at one. This
matches the covariance convention used by the statsmodels GLM path rather
than glum's optional sandwich or finite-sample corrections.
"""
def __init__(self, params, bse, pvalues, resid, fitted, scale,
df_model, df_resid, nobs, model, family, llf,
null_deviance, deviance, n_iter, llnull=None):
"""Store glum estimates in the statsmodels-compatible results shape.
Every argument is named because this class exists to be READ LIKE A
statsmodels RESULT, and a reader who cannot tell which of sixteen
positional values is the null deviance cannot check that claim.
:param params: fitted coefficients, a Series indexed by column name.
:param bse: their standard errors, from the canonical-link
information matrix described in the class docstring.
:param pvalues: two-sided p values for the coefficients.
:param resid: response residuals, ``y - mu``.
:param fitted: fitted values on the RESPONSE scale, ``mu``.
:param scale: the dispersion. One for the fixed-dispersion families
and the Pearson estimate for Gaussian, which is the convention
:mod:`spacr.regression_qc` reads off ``model.scale``.
:param df_model: model degrees of freedom, coefficients less the
intercept.
:param df_resid: residual degrees of freedom, rows less coefficients.
:param nobs: number of observations the fit used.
:param model: the design object, carrying the column names and
presented as a ``GLM`` class so the QC path resolves it.
:param family: the glum family, which supplies ``loglike`` and
``deviance``.
:param llf: log-likelihood of the fitted model.
:param null_deviance: deviance of the intercept-only model.
:param deviance: deviance of the fitted model.
:param n_iter: iterations the solver took, 0 when it does not report.
:param llnull: log-likelihood of the NULL model, optional.
CARRIED RATHER THAN DERIVED. ``fit_quality_note`` falls back to
``null_deviance / -2`` when it is absent, so a backend passing
only the deviance prints a different McFadden from statsmodels
for the identical fit -- the one thing this class exists not to
do. The null model is fitted anyway; this only keeps its answer.
"""
self.params = params
self.bse = bse
self.pvalues = pvalues
with np.errstate(divide='ignore', invalid='ignore'):
self.tvalues = params / bse
self.resid = resid
self.resid_response = resid
self.fittedvalues = fitted
self.scale = float(scale)
self.df_model = float(df_model)
self.df_resid = float(df_resid)
self.nobs = float(nobs)
self.model = model
self.family = family
self.llf = float(llf)
self.null_deviance = float(null_deviance)
self.llnull = None if llnull is None else float(llnull)
self.deviance = float(deviance)
self.n_iter = int(n_iter)
def predict(self, exog=None):
"""In-sample fitted values on the RESPONSE scale, as sm's GLM does."""
if exog is None:
return self.fittedvalues
raise ValueError(
"this GLM was fitted by glum through spaCR's design matrix and "
"does not carry the link's inverse for a new row. Fit with "
"regression_backend='statsmodels' if you need to predict.")
def summary(self):
"""Return the text-summary adapter for this fit."""
return _GlumSummary(self)
class _GlumSummary:
"""``.as_text()`` for :class:`_GlumResults`."""
def __init__(self, results):
"""Bind the glum results rendered by this summary.
:param results: the :class:`_GlumResults` to render. Held, not
copied; see :class:`_AbsorbedSummary`.
"""
self._results = results
def as_text(self):
"""Return a multiline report of the glum fit."""
r = self._results
lines = [
f"Generalized linear model fitted by glum "
f"({type(r.family).__name__})",
f" observations {int(r.nobs)}",
f" coefficients {len(r.params)}",
f" residual df {int(r.df_resid)}",
f" deviance {r.deviance:.6g}",
f" null deviance {r.null_deviance:.6g}",
f" log-likelihood {r.llf:.6g}",
f" IRLS steps {r.n_iter}",
"",
"coefficient / std err / z / P>|z|",
]
for name in r.params.index:
lines.append(f" {name} {r.params[name]:.6g} "
f"{r.bse[name]:.6g} {r.tvalues[name]:.4f} "
f"{r.pvalues[name]:.4g}")
return "\n".join(lines)
def __str__(self):
"""Return the same report as :meth:`as_text`."""
return self.as_text()
def _glum_information_weights(family, mu, var_weights):
"""``W_ii`` of the GLM information matrix, for the families glum fits.
For a canonical link ``dmu/deta`` equals the variance function, so the
weight collapses to ``v_i V(mu_i)``: ``v * mu`` for a log-link Poisson and
``v * mu (1 - mu)`` for a logit Binomial. Gaussian identity is ``v``. They
are written out per family rather than differenced numerically because a
finite difference here would put its own error into every standard error
on the volcano.
"""
weights = np.asarray(var_weights, dtype=float).reshape(-1)
mu = np.asarray(mu, dtype=float).reshape(-1)
if isinstance(family, sm.families.Poisson):
return weights * mu
if isinstance(family, sm.families.Binomial):
return weights * mu * (1.0 - mu)
if isinstance(family, sm.families.Gaussian):
return weights
raise ValueError(
f"spaCR does not know the information weight for "
f"{type(family).__name__}, so it cannot form the standard errors of a "
f"glum fit of it. Fit with regression_backend='statsmodels'.")
def _fit_glum_glm(X, y, regression_type, weights=None, exposure=None):
"""Fit one of the GLM families through glum instead of statsmodels.
Use glum's IRLS and active-set solver for supported Poisson, binomial, or
automatically selected GLM families. Small designs may not amortize the
backend's setup cost; its advantage is intended for wide model matrices.
:param X: the design DataFrame.
:param y: the response.
:param regression_type: one of :data:`_GLUM_FAMILIES`.
:param weights: per-well cell counts, used as ``var_weights`` by the
binomial families exactly as the statsmodels path uses them.
:param exposure: per-well cell counts for the Poisson ``offset(log(.))``.
:returns: :class:`_GlumResults`.
:raises ValueError: for a family glum cannot fit, or a response the family
refuses.
"""
columns = list(getattr(X, 'columns', []))
if not columns:
raise ValueError(
"the glum backend reports one coefficient per design column and "
"reads the names from the design, so it needs a DataFrame; a "
"bare array has no names.")
design = np.asarray(X, dtype=float)
y_flat = np.asarray(y, dtype=float).reshape(-1)
n = y_flat.size
offset = None
var_weights = np.ones(n, dtype=float)
if regression_type == 'poisson':
_validate_poisson_response(y, X)
family = sm.families.Poisson(link=sm.families.links.Log())
elif regression_type == 'logit':
family = sm.families.Binomial(link=sm.families.links.Logit())
else:
family = pick_glm_family_and_link(y)
if isinstance(family, sm.families.Poisson):
_validate_poisson_response(y, X)
if isinstance(family, sm.families.Poisson):
n_total = None
if exposure is not None:
n_total = np.asarray(exposure, dtype=float).reshape(-1)
if n_total.size != n:
raise ValueError(
f"the Poisson exposure has {n_total.size} entries but "
f"the response has {n}; each well must carry its own "
f"cell count.")
if not np.isfinite(n_total).all() or np.any(n_total <= 0):
raise ValueError(
"the Poisson exposure is the well's cell count, so it "
f"must be finite and strictly positive; got "
f"{np.nanmin(n_total)}-{np.nanmax(n_total)}.")
offset = np.log(n_total)
else:
print("Warning: no per-well cell count reached the Poisson fit, "
"so it models the raw count with no offset(log(cell_count)).")
elif isinstance(family, sm.families.Binomial) and weights is not None:
var_weights = np.asarray(weights, dtype=float).reshape(-1)
if var_weights.size != n:
raise ValueError(
f"the binomial fit was given {var_weights.size} weights for "
f"{n} observations.")
glum_family = {
'Poisson': 'poisson', 'Binomial': 'binomial', 'Gaussian': 'normal',
}.get(type(family).__name__)
if glum_family is None:
raise ValueError(
f"regression_backend='glum' cannot fit a "
f"{type(family).__name__} family; spaCR routes poisson, binomial "
f"and gaussian through it. Fit with "
f"regression_backend='statsmodels'.")
from glum import GeneralizedLinearRegressor
estimator = GeneralizedLinearRegressor(
family=glum_family, alpha=0, fit_intercept=False,
gradient_tol=1e-10, max_iter=500)
fit_kwargs = {}
if offset is not None:
fit_kwargs['offset'] = offset
if weights is not None and isinstance(family, sm.families.Binomial):
fit_kwargs['sample_weight'] = var_weights
estimator.fit(design, y_flat, **fit_kwargs)
beta = np.asarray(estimator.coef_, dtype=float).reshape(-1)
eta = design @ beta + (0.0 if offset is None else offset)
mu = family.link.inverse(eta)
info_w = _glum_information_weights(family, mu, var_weights)
xtwx = design.T @ (design * info_w[:, None])
if isinstance(family, sm.families.Gaussian):
scale = float(np.sum(info_w * (y_flat - mu) ** 2)) / (n - len(beta))
else:
scale = 1.0
try:
cov = scale * np.linalg.inv(xtwx)
except np.linalg.LinAlgError as exc:
raise ValueError(
f"the glum fit's information matrix is singular ({exc}), so its "
f"{len(beta)} coefficients are not identified. statsmodels "
f"answers the same design with a pseudo-inverse, which picks one "
f"arbitrary solution out of infinitely many.") from exc
se = np.sqrt(np.clip(np.diag(cov), 0.0, None))
with np.errstate(divide='ignore', invalid='ignore'):
z = np.where(se > 0, beta / se, 0.0)
p_values = 2.0 * st.norm.sf(np.abs(z))
null_kwargs = {'family': family}
if offset is not None:
null_kwargs['offset'] = offset
if weights is not None and isinstance(family, sm.families.Binomial):
null_kwargs['var_weights'] = var_weights
null_fit = sm.GLM(y_flat, np.ones((n, 1)), **null_kwargs).fit()
names = [str(c) for c in columns]
model = _AbsorbedDesign(y_flat, design, names)
model.__class__ = _GLUM_DESIGN_CLASS
llf = family.loglike(y_flat, mu, var_weights=var_weights, scale=scale)
deviance = family.deviance(y_flat, mu, var_weights=var_weights)
return _GlumResults(
params=pd.Series(beta, index=names),
bse=pd.Series(se, index=names),
pvalues=pd.Series(p_values, index=names),
resid=y_flat - mu, fitted=mu, scale=scale,
df_model=len(beta) - 1, df_resid=n - len(beta), nobs=n, model=model,
family=family, llf=llf, null_deviance=float(null_fit.null_deviance),
llnull=float(null_fit.llf),
deviance=deviance, n_iter=int(getattr(estimator, 'n_iter_', 0)))
#: The design class name :mod:`spacr.regression_qc` resolves a GLM by. Its
#: scale rule reads the dispersion off ``model.scale``, which is what
#: :class:`_GlumResults` reports -- 1 for the fixed-dispersion families and
#: the Pearson estimate for Gaussian, exactly as statsmodels does.
_GLUM_DESIGN_CLASS = type('GLM', (_AbsorbedDesign,),
{'__doc__': "The design of a glum-fitted GLM."})
[docs]
def regression_model(X, y, regression_type='ols', groups=None, alpha=1.0,
cov_type=None, weights=None, l1_ratio=0.5, quantile=0.5,
hinge_threshold=None, huber_t=1.345, exposure=None,
spline_knots=4, spline_degree=3,
group_lasso_lambda='auto', rra_alpha=0.25,
rra_permutations=10000,
regression_backend=DEFAULT_REGRESSION_BACKEND,
verbose=False, response_name="", transform="",
glm_force_identity=False):
"""Dispatch to the requested regression backend and return the fitted model.
Every name in :data:`REGRESSION_TYPES` is fittable here, and every one of
them has a matching branch in :func:`process_model_coefficients`, so a
model that fits can always be turned into a coefficient table.
The backends, and what each is for:
================================== ========================================================
``ols`` Ordinary least squares on a continuous well response.
``wls`` Weighted least squares; ``weights`` is the well's cell
count, so a well of 400 cells outweighs one of 30.
``rlm``/``huber`` Robust M-estimation (Huber loss). For outlier-heavy
wells: a handful of runaway wells no longer drag the
fit.
``glm`` GLM with the family auto-selected from the response by
:func:`pick_glm_family_and_link`.
``poisson`` Poisson GLM with a log link and
``offset(log(exposure))``, for per-well counts - so the
coefficients are effects on the per-cell RATE, not on
the well's headcount.
``quasi_binomial`` Binomial GLM whose dispersion is estimated from the
Pearson chi-square, for overdispersed fractions.
``beta`` Beta regression, for a fraction strictly inside (0, 1).
``logit``/``probit`` GLM-binomial on a fraction, weighted by cell count.
``quantile`` Quantile regression at ``quantile``; fits the tail of
the response rather than its mean.
``mixed`` Mixed-effects linear model with ``groups`` as the random
intercept.
``lasso``/``ridge``/``elasticnet`` Penalised least squares.
``hinge`` Linear SVM (hinge loss) on a binarised response.
``horseshoe`` Sparse Poisson GLM with a horseshoe prior (spaCRPower's
power-analysis model), via :mod:`spacr.power_model`.
``group_lasso`` Penalised least squares with a gene's guide columns
penalised as ONE block, so a gene is selected or dropped
as a set rather than one guide at a time - the penalised
analogue of the mixed model's nesting, via
:mod:`spacr.group_lasso`.
``rra`` MAGeCK-style robust rank aggregation: guides ranked by
their marginal effect, aggregated to the gene BY RANK
with a permutation P value, via :mod:`spacr.rra`. It
forms no joint fit, so the collinearity and the p >> n
width that constrain every backend above do not reach it.
================================== ========================================================
Settings a backend cannot read are REFUSED, not ignored — see
:data:`REGRESSION_SETTINGS_USED`.
:param X: Design matrix (DataFrame; column names become feature names).
:param y: Response variable.
:param regression_type: One of :data:`REGRESSION_TYPES`.
:param regression_backend: WHO fits it -- one of
:data:`REGRESSION_BACKEND_ORDER`. Default ``'statsmodels'``, which
produced every existing result. A backend that cannot fit
``regression_type`` is REFUSED here with the reason, not ignored:
the two controls constrain each other in both directions
, and a settings CSV reaches this function
without passing a panel that could have greyed the entry out.
:param groups: Cluster identifiers for the mixed model.
:param alpha: Penalty weight for ``lasso``/``ridge``/``elasticnet`` and
the inverse SVM margin for ``hinge``; ``'auto'`` / ``None`` picks it by
5-fold cross-validation for all four (mean squared error for the
penalised least-squares three, balanced accuracy for ``hinge``).
:param cov_type: Covariance estimator for the likelihood fits
(``'HC0'..'HC3'``); ``None`` for classical standard errors.
:param weights: Per-observation weights - the well's cell count. Used as
``var_weights`` by ``logit``/``probit``/``quasi_binomial`` and as the
WLS weights by ``wls``.
:param l1_ratio: ``elasticnet`` mix; 1.0 is lasso, 0.0 is ridge.
:param quantile: Quantile fitted by ``quantile`` regression, in (0, 1).
:param hinge_threshold: Cut used to binarise a continuous response for
``hinge``; see :func:`binarise_response`.
:param spline_knots: Knots per continuous covariate for ``spline``.
:param spline_degree: Polynomial degree of that basis; 3 is cubic.
:param huber_t: Huber tuning constant for ``rlm``/``huber``, in units of
the estimated residual scale. 1.345 gives 95% efficiency under
normality.
:param exposure: Per-observation exposure (the well's cell count) used as
``offset(log(exposure))`` by ``horseshoe`` and by ``poisson`` (and by
``glm`` when it auto-selects a Poisson family).
:param group_lasso_lambda: The block penalty for ``group_lasso``. Its own
key rather than ``alpha`` because it is compared against
:func:`spacr.group_lasso.max_lambda`, which is a property of the
design, so a value carried over from a lasso run would mean something
else here.
:param rra_alpha: The top fraction of the guide ranking alpha-RRA
aggregates over. MAGeCK's 0.25, which is what keeps a gene with one
strong guide and three that did not cut findable.
:param rra_permutations: Draws per distinct guide count in RRA's
permutation null; 10,000 puts the smallest reportable P value at 1e-4.
:returns: Fitted statsmodels / sklearn estimator.
:raises ValueError: on an unsupported ``regression_type``, or when a
setting the chosen backend cannot read was set to a non-default value.
Example:
.. code-block:: python
import pandas as pd
X = pd.DataFrame({'Intercept': 1.0, 'fraction': [0.1, 0.5, 0.9]})
model = regression_model(X, pd.Series([0.2, 0.4, 0.7]), 'ols')
model.params['fraction'] # the recovered slope
"""
if regression_type in UNSUPPORTED_REGRESSION_TYPES:
raise ValueError(
f"Unsupported regression type {regression_type}: "
f"{UNSUPPORTED_REGRESSION_TYPES[regression_type]}")
if regression_type not in REGRESSION_TYPES:
raise ValueError(
f"Unsupported regression type {regression_type}. "
f"Supported types: {list(REGRESSION_TYPES)}")
y_flat = np.asarray(y, dtype=float).reshape(-1)
use_auto_alpha = alpha is None or (isinstance(alpha, str) and alpha == 'auto')
if _left_blank(cov_type):
cov_type = None
if _left_blank(hinge_threshold):
hinge_threshold = None
supplied = {
'alpha': 1.0 if use_auto_alpha else alpha,
'l1_ratio': l1_ratio,
'cov_type': cov_type,
'quantile': quantile,
'hinge_threshold': hinge_threshold,
'huber_t': huber_t,
'spline_knots': spline_knots,
'spline_degree': spline_degree,
'group_lasso_lambda': group_lasso_lambda,
'rra_alpha': rra_alpha,
'rra_permutations': rra_permutations,
}
backend = _require_backend(regression_type, regression_backend)
_reject_unused_settings(regression_type, {
name: (supplied[name], default)
for name, default in _MODEL_LEVEL_DEFAULTS.items()})
def _find_best_alpha(model_cls):
"""Fit and return the requested cross-validated penalty estimator."""
alphas = np.logspace(-5, 5, 100)
if model_cls == 'lasso':
cv = LassoCV(alphas=alphas, cv=5, max_iter=10000).fit(X, y_flat)
elif model_cls == 'ridge':
cv = RidgeCV(alphas=alphas, cv=5).fit(X, y_flat)
elif model_cls == 'elasticnet':
cv = ElasticNetCV(alphas=alphas, l1_ratio=l1_ratio, cv=5,
max_iter=10000).fit(X, y_flat)
else:
raise ValueError(f"_find_best_alpha called with unknown model_cls={model_cls!r}")
print(f"Optimal alpha for {model_cls}: {cv.alpha_:.4g} "
f"(MSE: {mean_squared_error(y_flat, cv.predict(X)):.4f})")
return cv
def _glm_binomial(link=None, scale=None):
"""Fit and return an optionally weighted binomial GLM for ``link`` and ``scale``."""
family = sm.families.Binomial(link=link) if link else sm.families.Binomial()
kwargs = {'family': family}
if weights is not None:
kwargs['var_weights'] = np.asarray(weights).ravel()
fit_kwargs = {}
if scale is not None:
fit_kwargs['scale'] = scale
if cov_type is not None:
fit_kwargs['cov_type'] = cov_type
return sm.GLM(y, X, **kwargs).fit(**fit_kwargs)
def _poisson_offset():
"""``log(exposure)``, or None with a warning when there is no exposure.
A per-well POSITIVE COUNT is not comparable between wells of different
size: ``process_scores`` sums the response for the count models, so a
well of 2000 cells contributes roughly four times the count of a well
of 500 at the identical underlying rate. Modelling that count without
``offset(log(Ntotal))`` asks the covariates to explain well size, and
any covariate correlated with it comes back as a hit. Measured on a
400-well simulation with a nuisance covariate that drives well size and
nothing else, and a true rate coefficient of +1.5: without the offset
the nuisance term came back at +1.88 with p = 0, ahead of the real
effect; with it, +0.002 with p = 0.90.
:returns: ``log(exposure)`` aligned with ``y``, or None.
:raises ValueError: when the exposure is not positive and finite —
``log`` of it would be NaN/-inf and every downstream number would
silently follow.
"""
if exposure is None:
print("Warning: no per-well cell count reached the Poisson fit, so "
"it models the raw count with no offset(log(cell_count)). "
"Wells of different size are then not comparable and any "
"covariate correlated with well size will look like a hit. "
"Run the scores through process_scores so each well carries "
"its cell count.")
return None
n_total = np.asarray(exposure, dtype=float).ravel()
if n_total.size != np.asarray(y, dtype=float).reshape(-1).size:
raise ValueError(
f"the Poisson exposure has {n_total.size} entries but the "
f"response has {np.asarray(y).reshape(-1).size}; each well "
f"must carry its own cell count.")
if not np.isfinite(n_total).all() or np.any(n_total <= 0):
raise ValueError(
"the Poisson exposure is the well's cell count, so it must be "
f"finite and strictly positive; got "
f"{np.nanmin(n_total)}-{np.nanmax(n_total)}. A well with no "
f"cells has no rate to estimate and must be filtered out "
f"(min_cells_per_well) rather than offset by log(0).")
return np.log(n_total)
def _glm_auto():
"""Return a forced identity-link fit or the response-appropriate GLM."""
fit_y = y
if glm_force_identity:
family = sm.families.Gaussian(link=sm.families.links.Identity())
print(f" Using Gaussian family with Identity link for "
f"{response_name or 'the response'}.")
return sm.GLM(fit_y, X, family=family).fit(
**({'cov_type': cov_type} if cov_type else {}))
family = pick_glm_family_and_link(fit_y, name=response_name,
transform=transform)
if isinstance(family, sm.families.Poisson):
_validate_poisson_response(fit_y, X)
return sm.GLM(fit_y, X, family=family, offset=_poisson_offset()).fit(
**({'cov_type': cov_type} if cov_type else {}))
kwargs = {'family': family}
if weights is not None and isinstance(family, sm.families.Binomial):
kwargs['var_weights'] = np.asarray(weights).ravel()
return sm.GLM(fit_y, X, **kwargs).fit(
**({'cov_type': cov_type} if cov_type else {}))
def _glm_poisson():
"""Validate counts and return a Poisson-log fit with any exposure offset."""
_validate_poisson_response(y, X)
family = sm.families.Poisson(link=sm.families.links.Log())
return sm.GLM(y, X, family=family, offset=_poisson_offset()).fit(
**({'cov_type': cov_type} if cov_type else {}))
def _wls():
"""Validate captured per-well weights and return the fitted WLS model."""
if weights is None:
raise ValueError(
"regression_type='wls' needs per-well weights, and no "
"'cell_count' column reached the model. Weighted least "
"squares with unit weights is exactly OLS, so spaCR will not "
"fit it under the 'wls' label. Use 'ols', or run the scores "
"through process_scores so each well carries its cell count.")
w = np.asarray(weights, dtype=float).ravel()
if not np.isfinite(w).all() or np.any(w <= 0):
raise ValueError(
"WLS weights must be finite and positive (they are per-well "
f"cell counts); got {np.nanmin(w)}-{np.nanmax(w)}.")
return sm.WLS(y, X, weights=w).fit(
**({'cov_type': cov_type} if cov_type else {}))
def _rlm():
"""Return a robust linear fit using the captured Huber tuning constant."""
return sm.RLM(y, X, M=sm.robust.norms.HuberT(t=huber_t)).fit()
def _quantile():
"""Validate the captured quantile and return its fitted regression."""
if not 0.0 < float(quantile) < 1.0:
raise ValueError(
f"quantile must lie strictly inside (0, 1); got {quantile!r}. "
f"0.5 is the median fit.")
return sm.QuantReg(y, X).fit(q=float(quantile))
def _hinge():
"""Return a hinge classifier using auto-CV or fixed ``C = 1 / alpha``."""
y_binary = binarise_response(y, hinge_threshold,
name='dependent variable')
if use_auto_alpha:
return _find_best_hinge_alpha(y_binary)
strength = float(alpha)
if strength <= 0:
raise ValueError(
f"alpha must be positive for hinge regression; got {alpha!r}.")
model = _hinge_estimator(strength)
model.fit(X, y_binary)
return model
def _hinge_estimator(strength):
"""A LinearSVC at regularisation ``strength`` (``C = 1 / strength``).
``class_weight='balanced'`` because a screen's positive class is
routinely a small minority of wells: an unweighted hinge on a 95/5
split minimises its loss by calling every well negative, which returns
a coefficient vector of ~0 for every gRNA and reads downstream as "no
hits". Balancing reweights each class by its inverse frequency, so the
decision boundary is fitted to separate the classes rather than to
count them.
"""
return LinearSVC(C=1.0 / strength, loss='hinge', dual=True,
max_iter=20000, random_state=0,
class_weight='balanced')
def _find_best_hinge_alpha(y_binary):
"""Pick the hinge penalty by stratified CV on balanced accuracy.
The same 5-fold shape ``_find_best_alpha`` uses for the penalised
least-squares backends. Balanced accuracy rather than accuracy: on an
imbalanced screen plain accuracy is maximised by the degenerate
all-negative fit, so scoring on it would cross-validate its way to the
very failure ``class_weight='balanced'`` exists to prevent.
Falls back to the unpenalised-scale default ``C = 1`` when the response
has too few wells in a class to split five ways — a two-fold CV on
three positive wells is noise, and choosing a penalty from noise is
worse than not choosing one.
"""
from sklearn.model_selection import cross_val_score
strengths = np.logspace(-3, 3, 13)
minority = int(min(np.sum(y_binary == 0), np.sum(y_binary == 1)))
n_splits = min(5, minority)
if n_splits < 2:
print(f"hinge: alpha='auto' needs at least two wells in each "
f"class to cross-validate and the smaller class has "
f"{minority}; falling back to alpha=1.")
model = _hinge_estimator(1.0)
model.fit(X, y_binary)
return model
folds = StratifiedKFold(n_splits=n_splits, shuffle=True,
random_state=0)
scores = []
for strength in strengths:
with warnings.catch_warnings():
warnings.simplefilter('ignore')
fold_scores = cross_val_score(
_hinge_estimator(strength), X, y_binary, cv=folds,
scoring='balanced_accuracy')
scores.append(float(np.mean(fold_scores)))
best = float(strengths[len(strengths) - 1
- int(np.argmax(scores[::-1]))])
print(f"Optimal alpha for hinge: {best:.4g} "
f"(balanced accuracy {max(scores):.4f}, {n_splits}-fold)")
model = _hinge_estimator(best)
model.fit(X, y_binary)
return model
def _named_design(name):
"""The design's column names, or a refusal that says why they matter.
Gene-aware backends recover the gene behind each predictor from the
column name produced by patsy. Refuse an unnamed array because group
lasso would otherwise treat every column as a separate gene and
reduce to ordinary lasso under a different label.
"""
columns = getattr(X, 'columns', None)
if columns is None:
raise ValueError(
f"regression_type={name!r} groups the design's columns by "
f"gene and reads that grouping from the COLUMN NAMES, so it "
f"needs a DataFrame design; a bare array has no names to "
f"group by. Build the design with "
f"dmatrices(..., return_type='dataframe'), which is what the "
f"pipeline hands in.")
return columns
def _group_lasso():
"""Fit the named gene-grouped design and return compatible results."""
from . import group_lasso as group_lasso_module
columns = _named_design('group_lasso')
design = np.asarray(X, dtype=float)
blocks = _design_column_groups(columns)
gene_terms = _level_term_mask(columns)
if _left_blank(group_lasso_lambda) or (
isinstance(group_lasso_lambda, str)
and group_lasso_lambda.strip().lower() == 'auto'):
lam = group_lasso_module.choose_lambda(
design, y_flat, blocks,
required=gene_terms if gene_terms.any() else None)
print(f"group_lasso_lambda='auto': cross-validated over "
f"{group_lasso_module.PATH_POINTS} penalties down from "
f"this design's ceiling of "
f"{group_lasso_module.max_lambda(design, y_flat, blocks):.4g}"
f", chose {lam:.4g}.")
else:
lam = float(group_lasso_lambda)
beta, intercept, converged = group_lasso_module.fit(
design, y_flat, blocks, lam=lam)
if not gene_terms.any():
raise ValueError(
"regression_type='group_lasso' penalises a GENE's guide "
f"columns as one block, and none of this design's "
f"{len(gene_terms)} columns is a gRNA or gene term "
f"(columns: {[str(c) for c in columns][:6]}). Every column "
f"would be its own block, which is ordinary lasso under "
f"another name. It is fitted on the design prepare_formula "
f"builds, whose terms are 'fraction:grna[...]' or "
f"'gene_fraction:gene[...]'.")
if not np.any(beta[gene_terms]):
ceiling = group_lasso_module.max_lambda(design, y_flat, blocks)
raise ValueError(
f"group_lasso shrank every one of the "
f"{int(gene_terms.sum())} gRNA/gene coefficients to exactly "
f"zero at group_lasso_lambda={lam!r}, so the fit carries no "
f"information about any gene. This design's "
f"group_lasso.max_lambda -- the penalty above which nothing "
f"at all survives -- is {ceiling:.4g}, and the gene blocks "
f"empty well below it. Set group_lasso_lambda='auto' to "
f"cross-validate it, or a small fraction of that ceiling to "
f"choose it yourself. Or fit an unpenalised model ('ols') to "
f"see the effect sizes the penalty is shrinking away.")
if not converged:
print(f"Warning: the group lasso did not reach its tolerance in "
f"{group_lasso_module.MAX_ITERATIONS} sweeps. The "
f"coefficients are the last iterate, not the solution; "
f"treat the selection as provisional.")
genes_in_design = {label for label, is_gene in zip(blocks, gene_terms)
if is_gene}
selected = {label for label, coefficient, is_gene
in zip(blocks, beta, gene_terms)
if is_gene and coefficient != 0}
model = _GroupLassoResults(beta, intercept, blocks, lam, converged)
mse = mean_squared_error(y_flat, model.predict(X))
print(f"Group lasso MSE: {mse:.4f}, lambda={lam:g}, "
f"{len(selected)} of {len(genes_in_design)} gene blocks "
f"selected ({int(np.sum(beta[gene_terms] != 0))} of "
f"{int(gene_terms.sum())} gRNA columns).")
return model
def _rra():
"""Rank marginal design slopes and return gene-level alpha-RRA results."""
from . import rra as rra_module
columns = _named_design('rra')
design = np.asarray(X, dtype=float)
genes = [_gene_of_design_column(column) for column in columns]
gene_terms = _level_term_mask(columns)
if not gene_terms.any():
raise ValueError(
"regression_type='rra' aggregates a GENE's guides by rank, "
f"and none of this design's {len(genes)} columns is a "
f"gRNA or gene term (columns: {[str(c) for c in columns][:6]}"
f"). It is fitted on the design prepare_formula builds, "
f"whose terms are 'fraction:grna[...]' or "
f"'gene_fraction:gene[...]'.")
centred = design - design.mean(axis=0)
response = y_flat - float(y_flat.mean())
spread = (centred ** 2).sum(axis=0)
moving = spread > 0
slopes = np.zeros(design.shape[1], dtype=float)
slopes[moving] = (centred[:, moving].T @ response) / spread[moving]
ranked = np.where(moving, slopes, np.nan)
table = rra_module.rank_aggregate(
ranked, genes, alpha=float(rra_alpha), direction='both',
n_permutations=int(rra_permutations))
if not len(table) or 'p_neg' not in table.columns:
raise ValueError(
"regression_type='rra' ranked no guide: every gRNA column of "
"this design is constant, so no guide has a marginal effect "
"to rank. Check the fraction threshold - a design whose guide "
"columns do not vary carries no information about any guide.")
two_sided = np.minimum(1.0, 2.0 * np.minimum(
table['p_neg'].to_numpy(dtype=float),
table['p_pos'].to_numpy(dtype=float)))
by_gene = dict(zip(table['gene'].astype(str), two_sided))
p_values = np.array(
[by_gene.get(str(gene), np.nan) if gene is not None else np.nan
for gene in genes], dtype=float)
called = int(np.sum(two_sided <= 0.05))
print(f"RRA: {len(table)} genes aggregated from "
f"{int(np.sum(moving & gene_terms))} ranked guides, "
f"alpha={float(rra_alpha):g}, {int(rra_permutations)} "
f"permutations per guide count; {called} genes at an "
f"uncorrected two-sided p <= 0.05.")
return _RRAResults(slopes, p_values, columns, table)
def _horseshoe():
"""Return a horseshoe-Poisson fit for the captured design and exposure."""
return _fit_horseshoe_poisson(X, y, exposure)
def _spline():
"""OLS on a design whose COVARIATES carry a spline basis.
The guide columns are untouched, so one coefficient and one P value
per guide survive. The volcano, hit list and attribution can therefore
read the result with no special case. What becomes free to bend is the
nuisance trend that the straight line was assuming away.
A column is treated as a covariate when it is CONTINUOUS and is not
a guide or gene term -- an indicator has nothing to bend through,
and expanding one would spend degrees of freedom on nothing.
"""
from .nonparametric_fits import spline_design
covariates = []
for name in getattr(X, "columns", []):
label = str(name)
if "grna[" in label or "gene[" in label or label == "Intercept":
continue
column = np.asarray(X[name], dtype=float)
if np.unique(column).size > 4:
covariates.append(name)
design = (spline_design(X, covariates,
knots=int(spline_knots),
degree=int(spline_degree))
if covariates else X)
fitted = (sm.OLS(y, design).fit(cov_type=cov_type) if cov_type
else sm.OLS(y, design).fit())
return fitted
model_map = {
'ols': lambda: sm.OLS(y, X).fit(cov_type=cov_type) if cov_type else sm.OLS(y, X).fit(),
'spline': _spline,
'wls': _wls,
'rlm': _rlm,
'huber': _rlm,
'glm': _glm_auto,
'poisson': _glm_poisson,
'quasi_binomial': lambda: _glm_binomial(link=sm.families.links.Logit(),
scale='X2'),
'beta': lambda: BetaModel(endog=y, exog=X).fit(),
'logit': lambda: _glm_binomial(link=sm.families.links.Logit()),
'probit': lambda: _glm_binomial(link=sm.families.links.probit()),
'quantile': _quantile,
'mixed': lambda: perform_mixed_model(
y, X, groups, regression_backend=regression_backend),
'lasso': lambda: _find_best_alpha('lasso') if use_auto_alpha
else Lasso(alpha=alpha, max_iter=10000).fit(X, y_flat),
'ridge': lambda: _find_best_alpha('ridge') if use_auto_alpha
else Ridge(alpha=alpha).fit(X, y_flat),
'elasticnet': lambda: _find_best_alpha('elasticnet') if use_auto_alpha
else ElasticNet(alpha=alpha, l1_ratio=l1_ratio,
max_iter=10000).fit(X, y_flat),
'hinge': _hinge,
'horseshoe': _horseshoe,
'group_lasso': _group_lasso,
'rra': _rra,
}
if backend == 'pyfixest':
if cov_type is not None:
raise ValueError(
f"regression_backend='pyfixest' absorbs rowID and columnID, "
f"and the HC1/HC2/HC3 corrections are computed from the full "
f"model's leverage, which an absorbed fit never forms. It "
f"reports classical standard errors only, so "
f"cov_type={cov_type!r} would be a label on numbers that did "
f"not come from it. Fit with "
f"regression_backend='statsmodels' to use cov_type, or clear "
f"cov_type to absorb.")
if regression_type == 'wls' and weights is None:
raise ValueError(
"regression_type='wls' needs per-well weights, and no "
"'cell_count' column reached the model. Weighted least "
"squares with unit weights is exactly OLS, so spaCR will not "
"fit it under the 'wls' label. Use 'ols', or run the scores "
"through process_scores so each well carries its cell count.")
try:
model = _fit_absorbed_least_squares(
X, y, weights=weights if regression_type == 'wls' else None,
kind='WLS' if regression_type == 'wls' else 'OLS')
except ValueError as nothing_to_absorb:
if 'nothing to absorb' not in str(nothing_to_absorb):
raise
print(" regression_backend='pyfixest' has nothing to absorb "
"on this design -- there are no rowID or columnID terms in "
"it, so either model_plate_position=False removed them or "
"the screen sits on one row and one column. Fitting with "
"statsmodels instead: with no factors to project out the "
"two backends compute the same numbers, so this is the "
"same fit by the only route left, not a different model.")
model = model_map[regression_type]()
elif backend == 'glum':
if cov_type is not None:
raise ValueError(
f"regression_backend='glum' reports the classical GLM "
f"standard errors -- the inverse information matrix at a "
f"fixed dispersion -- and has no HC0..HC3 estimator that "
f"matches statsmodels', so cov_type={cov_type!r} would be a "
f"label on numbers that did not come from it. Fit with "
f"regression_backend='statsmodels' to use cov_type.")
model = _fit_glum_glm(X, y, regression_type, weights=weights,
exposure=exposure)
else:
model = model_map[regression_type]()
if regression_type in ['glm', 'poisson']:
print(fit_quality_note(model))
print(summary_for_console(model, verbose=verbose))
if regression_type in ['lasso', 'ridge', 'elasticnet']:
mse = mean_squared_error(y_flat, model.predict(X))
coefs = np.asarray(model.coef_).ravel()
n_nonzero = int(np.sum(coefs != 0))
print(f"{regression_type.capitalize()} regression MSE: {mse:.4f}, "
f"non-zero coefficients: {n_nonzero} of {X.shape[1]}")
if n_nonzero == 0:
if use_auto_alpha:
raise ValueError(
f"{regression_type} with alpha='auto' cross-validated its "
f"way to the empty model: every one of the "
f"{X.shape[1]} coefficients is exactly zero, because no "
f"gRNA predicted the held-out wells better than their mean "
f"did. That is a null screen, not a misconfiguration - the "
f"fit is refused rather than written out as '0 significant "
f"gRNAs', which is what it would look like. Check the "
f"dependent variable and the aggregation, or fit an "
f"unpenalised model ('ols') to see the effect sizes the "
f"penalty is shrinking away.")
raise ValueError(
f"{regression_type} shrank all {X.shape[1]} coefficients to "
f"exactly zero at alpha={alpha!r}: the penalty is far larger "
f"than the scale of this design, so the fit carries no "
f"information about any gRNA. Lower alpha, or set it to "
f"'auto' to choose it by cross-validation.")
return model
def _fit_horseshoe_poisson(X, y, exposure):
"""Fit spaCRPower's sparse Poisson model through :mod:`spacr.power_model`.
The model is the one ``spaCRPower/R/fit_model.R`` fits::
Npositive_w ~ Poisson(Ntotal_w * exp(b0 + sum_g b_g * log10expression_wg))
b_g ~ horseshoe(df = 10)
i.e. a Poisson GLM with a log link, an ``offset(log(Ntotal))`` exposure and
a horseshoe sparsity prior doing the variable selection. In spaCR's terms
``y`` is the per-well positive-object count (``process_scores`` sums the
response for this type, as it does for ``'poisson'``), ``exposure`` is the
well's cell count and ``X`` is the ordinary spaCR design.
The import is deliberately lazy and inside the branch: the horseshoe
fitter is a separate module, and neither the ordinary regressions nor
anything else that imports :mod:`spacr.ml` should pay for it or fail
without it.
:param X: Design matrix.
:param y: Per-well positive counts.
:param exposure: Per-well total cell counts (the Poisson exposure).
:returns: The fitted object returned by
``spacr.power_model.fit_horseshoe_poisson``, which must expose
``params`` and either ``pvalues`` or ``bse`` indexed like
``X.columns``.
:raises ImportError: when :mod:`spacr.power_model` is not installed yet,
naming the entry point this branch calls.
:raises ValueError: when no exposure is available, or the returned object
does not carry the coefficients this pipeline needs.
"""
if exposure is None:
raise ValueError(
"regression_type='horseshoe' fits Npositive ~ ... + "
"offset(log(Ntotal)), so it needs the per-well cell count as the "
"exposure, and no 'cell_count' column reached the model. Without "
"it the counts of a 400-cell well and a 30-cell well would be "
"compared as if the wells were the same size.")
try:
from .power_model import ModelData, fit_model, gather_model_estimate
except ImportError as exc:
raise ImportError(
"regression_type='horseshoe' needs spacr.power_model, which is "
"not present in this install. The branch calls "
"spacr.power_model.prepare/ModelData + fit_model + "
"gather_model_estimate; install or restore that module to use it."
) from exc
counts = _validate_poisson_response(
y, X, model="horseshoe (a sparse Poisson GLM over well counts)")
n_total = np.asarray(exposure, dtype=float).ravel()
if n_total.size != counts.size:
raise ValueError(
f"horseshoe exposure has {n_total.size} entries but the response "
f"has {counts.size}; they are the same wells and must align.")
if not np.isfinite(n_total).all() or np.any(n_total <= 0):
raise ValueError(
"horseshoe exposure (the well cell count) must be finite and "
f"positive; got {np.nanmin(n_total)}-{np.nanmax(n_total)}. "
"log(Ntotal) is undefined otherwise.")
if np.any(counts > n_total):
raise ValueError(
"horseshoe needs Npositive <= Ntotal per well: the response is a "
"count of positive objects and the exposure is how many objects "
"were imaged, so a well cannot have more positives than cells. "
f"{int(np.sum(counts > n_total))} well(s) break that.")
design = np.asarray(X, dtype=float)
columns = [str(c) for c in X.columns]
constant = [name for name, column in zip(columns, design.T)
if np.ptp(column) == 0]
model_data = ModelData(
wells=np.asarray(X.index),
genes=np.asarray(columns, dtype=object),
Npositive=counts,
Ntotal=n_total,
log10expression=design,
unidentified_genes=tuple(constant),
)
fit = fit_model(model_data, seed=0, standardize=True)
return _HorseshoeResults(fit, gather_model_estimate(fit))
class _HorseshoeResults:
"""Adapt a :class:`spacr.power_model.PowerFit` to the results API spaCR reads.
:func:`process_model_coefficients` wants ``params`` and ``pvalues``
indexed by design column; the horseshoe model reports posterior draws.
The translation is stated rather than implied:
* ``params`` is the posterior MEAN of each coefficient - a point estimate
under shrinkage, not a maximum-likelihood one, so it is already pulled
towards zero for terms the prior judges null. That is the whole purpose
of the model and the reason its coefficients are not comparable in
magnitude with the OLS ones.
* ``pvalues`` is the two-sided posterior TAIL MASS,
``2 * min(P(beta > 0), P(beta < 0))``. It is not a frequentist p-value
and no null hypothesis was tested to get it; it is reported under that
name because every consumer downstream - the volcano plot, the hit
table, ``-log10(p_value)`` - reads that column and would otherwise be
given nothing. A term whose posterior sits entirely on one side of zero
gets 0.
* Unidentified terms (a constant column, or a gRNA present in every well
at the same fraction) are DROPPED rather than reported as zero: the
model could not estimate them, and a zero with a p-value would read as
a tested null.
:param fit: the ``PowerFit`` returned by ``power_model.fit_model``.
:param estimates: the frame ``power_model.gather_model_estimate`` builds.
"""
def __init__(self, fit, estimates):
"""Validate the posterior table and expose identified coefficients."""
required = ('gene', 'mean', 'sd', 'prob_positive', 'identified')
missing = [c for c in required if c not in estimates.columns]
if missing:
raise ValueError(
f"spacr.power_model.gather_model_estimate returned columns "
f"{list(estimates.columns)}; spaCR's coefficient table needs "
f"{list(required)} and {missing} are absent.")
self.fit = fit
self.estimates = estimates
identified = estimates[estimates['identified'].astype(bool)]
index = pd.Index(identified['gene'].astype(str), name=None)
self.params = pd.Series(identified['mean'].to_numpy(), index=index)
self.bse = pd.Series(identified['sd'].to_numpy(), index=index)
prob_positive = identified['prob_positive'].to_numpy(dtype=float)
tail = 2.0 * np.minimum(prob_positive, 1.0 - prob_positive)
self.pvalues = pd.Series(np.clip(tail, 0.0, 1.0), index=index)
self.converged = bool(getattr(fit, 'converged', True))
if not self.converged:
print("Warning: the horseshoe fit did not meet its own "
"convergence criterion; treat the coefficients as "
"provisional and re-run with more steps or a NUTS backend.")
def summary(self):
"""Return the per-term posterior summary, for save_summary_to_file."""
return self.estimates
class _GroupLassoResults:
"""Adapt :mod:`spacr.group_lasso` to the estimator API spaCR reads.
``coef_`` and ``predict`` are all :data:`_SKLEARN_COEF_TYPES`' branch of
:func:`process_model_coefficients` and :func:`calculate_p_values` ask of a
penalised fit, so the group lasso reports EXACTLY what ``lasso`` and
``elasticnet`` report -- one signed coefficient per design column, and a
selection frequency attached by the run -- rather than a second convention
of its own.
WHY THE COEFFICIENT AND NOT ``gene_effects``' NORM. ``gene_effects``
answers with ``||b_g||_2``, one non-negative number per gene, which is the
natural summary of a block but is not what the pipeline downstream of the
fit is built on: the volcano's x axis, ``coefficient_threshold``'s
control spread and the hit table's sign all read a SIGNED per-column
effect. The block's own coefficients carry that sign, and because the
block is zero or none of it is, ``||b_g||_2 > 0`` and "this gene has a
non-zero coefficient" are the same statement -- so nothing is lost by
tabling the coefficients, and a caller who wants the norm is one
``np.linalg.norm`` away from it: ``coef_`` and ``groups`` are both on this
object, which is why ``groups`` is kept rather than discarded after the
fit.
:param coefficients: one coefficient per design column, in column order.
:param intercept: the unpenalised intercept.
:param groups: the group label of each column, from
:func:`_design_column_groups`.
:param lam: the penalty that was applied.
:param converged: whether block coordinate descent met its tolerance.
"""
def __init__(self, coefficients, intercept, groups, lam, converged):
"""Store flattened coefficients and fitted group-lasso metadata."""
self.coef_ = np.asarray(coefficients, dtype=float).ravel()
self.intercept_ = float(intercept)
self.groups = list(groups)
self.lam = float(lam)
self.converged = bool(converged)
def predict(self, X):
"""``X @ coef_ + intercept_``, the fit's prediction for a design.
:func:`calculate_p_values` needs the residual, and the residual needs
this. Taking ``np.asarray`` rather than relying on pandas' matmul keeps
it working for a plain array as well as the DataFrame the pipeline
hands in.
"""
return np.asarray(X, dtype=float) @ self.coef_ + self.intercept_
class _RRAResults:
"""Adapt :mod:`spacr.rra` to the results API :func:`process_model_coefficients` reads.
``params`` and ``pvalues``, the same two attributes every statsmodels fit
and :class:`_HorseshoeResults` expose, so ``rra`` needs no branch of its
own. What each one IS, stated rather than implied, because neither comes
from a joint fit:
* ``params`` is the guide's MARGINAL least-squares slope -- the slope of
the response on that guide's column alone, one parameter estimated at a
time. RRA's whole claim is that it never forms the joint fit (see
:mod:`spacr.rra`), and this screen is 823 guides against 610 wells,
where the joint fit is undefined. A marginal slope is defined for every
column at any width and is the analogue of MAGeCK's per-guide log fold
change, which is what alpha-RRA ranks.
* ``pvalues`` is the guide's GENE's permutation P value, two-sided as
``min(1, 2 * min(p_neg, p_pos))`` -- the standard combination of the two
one-sided permutation tests :func:`spacr.rra.rank_aggregate` reports.
Taking whichever tail is smaller and NOT doubling would be a one-sided
test chosen after seeing the data.
A row that names no gene -- the intercept, the row/column dummies -- was
never ranked, so its P value is NaN rather than 1.0: it was not tested, and
a 1.0 would read as "tested and found null".
:param scores: the marginal slope of each design column, in column order.
:param p_values: the two-sided permutation P value of each column's gene.
:param index: the design column names.
:param genes: :func:`spacr.rra.rank_aggregate`'s per-gene table, kept whole
so ``rho_neg``/``rho_pos`` and the direction split are not lost.
"""
def __init__(self, scores, p_values, index, genes):
"""Index marginal scores and permutation p-values by design column."""
feature_index = pd.Index([str(name) for name in index])
self.params = pd.Series(np.asarray(scores, dtype=float),
index=feature_index)
self.pvalues = pd.Series(np.asarray(p_values, dtype=float),
index=feature_index)
self.genes = genes
def summary(self):
"""The per-gene RRA table, for :func:`save_summary_to_file`."""
return self.genes
#: Regression types ``random_row_column_effects=True`` may be combined with.
#: ``'ols'`` is here because it is the DEFAULT value of ``regression_type``, so
#: it cannot be told apart from "the user never touched the model dropdown";
#: ``None`` means "choose from the response", which the mixed branch answers.
#: Every other name is a deliberate choice that the mixed override would throw
#: away.
_RANDOM_EFFECTS_COMPATIBLE = (None, 'ols', 'mixed')
def _reject_unused_run_settings(settings):
"""Refuse a post-fit setting the chosen model will never read.
:func:`regression_model` polices the six knobs that reach the estimator,
and raises before a wrong number can become a result. These three do not
reach the estimator at all — they configure how :func:`perform_regression`
turns coefficients into a hit list — so nothing was checking them, and
``lasso_selection_threshold=0.9`` on an OLS run passed through fifteen of
the seventeen types in silence.
``regression_type=None`` is policed as strictly as a named one:
:func:`check_distribution` only ever auto-selects ``logit``, ``beta``,
``quasi_binomial``, ``ols`` or ``glm``, none of which reads any of these,
so "it might pick lasso" is not a reason to let them through.
:param settings: The finished settings dict.
:raises ValueError: naming the setting, the type and the alternative.
"""
reg_type = settings.get('regression_type', 'ols')
_reject_unused_settings(reg_type, {
name: (settings.get(name, default), default)
for name, default in _RUN_LEVEL_DEFAULTS.items()})
return settings
def _reconcile_random_row_column_effects(settings):
"""Make ``random_row_column_effects=True`` and ``regression_type`` agree.
:func:`regression` reacts to the flag by fitting a MixedLM with row and
column variance components, whatever ``regression_type`` says — and
``_perform_regression_set_paths`` had already named the results folder
after ``settings['regression_type']``. A run configured as ``'lasso'`` with
the flag on therefore fitted a mixed model and wrote it to
``results/<screen>/lasso/``, where nothing in the folder, the volcano
filename or the settings CSV disagreed. Every penalty setting that run
carried was ignored too, silently, because the mixed branch never reaches
:func:`regression_model` and so never reaches
:func:`_reject_unused_settings`.
Two things happen here, both before any file is written:
* an incompatible model choice is REFUSED, naming both settings;
* a compatible one is rewritten to ``'mixed'`` in ``settings``, so the
folder, the volcano filename and the saved settings all name the model
that was actually fitted.
:param settings: The finished settings dict; mutated in place.
:raises ValueError: when the flag is combined with a named model that is
not a mixed model, with ``model_plate_position=False``, or with a
setting the mixed model cannot read.
"""
if not settings.get('random_row_column_effects', False):
return settings
if not settings.get('model_plate_position', True):
raise ValueError(
"random_row_column_effects=True fits rowID and columnID as "
"variance components, and model_plate_position=False takes them "
"out of the model entirely: there is nothing left for the mixed "
"fit to make random. Set model_plate_position=True to fit plate "
"position as variance components, or "
"random_row_column_effects=False to leave it out.")
reg_type = settings.get('regression_type', 'ols')
if reg_type not in _RANDOM_EFFECTS_COMPATIBLE:
raise ValueError(
f"random_row_column_effects=True fits a mixed model with row and "
f"column variance components, so it cannot also fit "
f"regression_type={reg_type!r}: one of the two has to go. It used "
f"to win silently, and the {reg_type!r} settings went with it — "
f"the run wrote a MixedLM fit into results/<screen>/{reg_type}/ "
f"and said nothing. Set random_row_column_effects=False to fit "
f"{reg_type!r}, or regression_type='mixed' to fit the mixed model.")
_reject_unused_settings('mixed', {
'alpha': (1.0 if settings.get('alpha') in (None, 'auto')
else settings.get('alpha', 1.0), 1.0),
'l1_ratio': (settings.get('l1_ratio', 0.5), 0.5),
'cov_type': (settings.get('cov_type'), None),
'quantile': (settings.get('quantile', 0.5), 0.5),
'hinge_threshold': (settings.get('hinge_threshold'), None),
'huber_t': (settings.get('huber_t', 1.345), 1.345),
'spline_knots': (settings.get('spline_knots', 4), 4),
'spline_degree': (settings.get('spline_degree', 3), 3),
})
if reg_type != 'mixed':
print(f"random_row_column_effects=True: fitting 'mixed' rather than "
f"{reg_type!r}, and naming the results folder for it.")
settings['regression_type'] = 'mixed'
return settings
def _mixed_model_groups(df, dependent_variable, model_index, *,
gene_column='gene'):
"""Return the outer random-intercept grouping for ``regression_type='mixed'``.
The gene is the outer cluster and the guide is nested inside it. The model
fits
y ~ gene_fraction:gene + (1 | gene/grna) + rowID + columnID
where ``(1 | gene/grna)`` is represented as ``groups=gene`` plus a guide
variance component inside each gene. Row and column structure is carried
by fixed terms, or by variance components when requested. See
:func:`fit_mixed_model`.
A single gene provides only one outer cluster and is refused.
:param df: The cleaned long-format frame.
:param dependent_variable: Response column name, named in the refusal so
the message points at the run the user actually configured.
:param model_index: Row index patsy kept, so the returned vector aligns
with the design matrix row for row.
:param gene_column: The outer grouping column. Default ``'gene'``.
:returns: Series of gene ids, one per design row.
:raises ValueError: when the screen has a single gene, naming the way out.
"""
genes = df.loc[model_index, gene_column]
n_genes = genes.nunique()
if n_genes > 1:
print(f"Mixed model: grouping on {gene_column} ({n_genes} genes), "
f"with guides nested inside. The gene sits above the guide, "
f"which is the level the random effect describes.")
return genes
raise ValueError(
f"a mixed model needs at least two clusters and this screen has one "
f"{gene_column}. The random intercept has to sit above the guide, so "
f"with a single gene there is nothing left for it to describe and "
f"every guide BLUP against {dependent_variable!r} would be shrunk to "
f"the same number. Fit a fixed-effects regression_type with "
f"level='grna', which tests the guides directly and is the model a "
f"one-gene screen supports.")
def _write_regression_qc(model, X, y, df, dst, *, coef_df=None,
regression_type=None, volcano_path=None):
"""Write the full QC suite for a fit into ``<dst>/regression_qc/``.
Generate residual, scale-location, Q-Q, influence, collinearity,
calibration, and coefficient diagnostics while the fitted design matrix
and response are still available.
Weights are deliberately not forwarded. ``regression`` passes cell
counts to ``regression_model`` as ``var_weights`` / WLS weights / Poisson
exposure for the types that take them, and for those types
:func:`spacr.regression_qc.build_context` recovers the weights from the
fitted model itself, so the hat diagonal, the residual and the scale agree.
The unweighted types (ols, lasso, ridge, elasticnet) never saw the counts,
and handing them in as ``weights`` would compute a weighted leverage for a
fit nobody ran. The counts still reach the cell-count panel through
``metadata['cell_count']``, which is where that panel looks first.
:param model: the fitted model.
:param X: the design matrix that was fitted.
:param y: the response that was fitted.
:param df: the cleaned long-format frame, for the per-well metadata.
:param dst: the run's results folder.
:param coef_df: the coefficient table, so the p-value histogram shows the
screen's p-values rather than the design's.
:param regression_type: the spaCR regression type string.
:param volcano_path: the volcano plot for this run, named on the report.
:returns: the manifest dict, or ``None`` if the report could not be written.
"""
from .regression_qc import regression_qc_report
metadata = None
try:
columns = [column for column in (schema.PLATE_KEY, schema.ROW_KEY,
schema.COLUMN_KEY, schema.PRC_KEY,
'cell_count')
if column in df.columns]
if columns:
metadata = df.loc[X.index, columns]
if len(metadata) != len(X):
raise ValueError(
f"{len(metadata)} metadata rows for {len(X)} fitted rows; "
f"the frame's index does not identify wells uniquely")
except Exception as error: # noqa: BLE001 - advisory
print(f"Regression QC: could not align per-well metadata to the fitted "
f"rows ({type(error).__name__}: {error}); the plate/row/column "
f"panels will skip rather than label the wrong well.")
metadata = None
try:
return regression_qc_report(
model, X, y, dst, metadata=metadata, coef_df=coef_df,
regression_type=regression_type, volcano_path=volcano_path,
verbose=True)
except Exception as error: # noqa: BLE001 - advisory
print(f"Regression QC report could not be written: "
f"{type(error).__name__}: {error}")
return None
[docs]
def resolve_levels(regression_type, level='both'):
"""Which level(s) a run fits, given the backend and the ``level`` setting.
``mixed`` fits ONE model that already contains both levels -- the gene as a
fixed effect and the guide as a random effect nested inside it -- so it
ignores ``level`` entirely and answers ``('gene',)``. That is why the GUI
greys the dropdown out rather than hiding it: the
setting exists, but this model does not read it.
Every other backend is fixed effects only and cannot nest, so it fits one
level at a time and ``level`` chooses which. ``'both'`` is TWO FITS.
:param regression_type: the backend name, or ``None`` (not yet chosen).
:param level: ``'both'`` (default), ``'grna'`` or ``'gene'``.
:returns: a tuple of levels to fit, in the order they are fitted.
:raises ValueError: for a level that is not one of :data:`LEVEL_CHOICES`.
"""
key = str(level).strip().lower()
if key not in LEVEL_CHOICES:
raise ValueError(
f"level={level!r} is not a model level. Choose one of "
f"{LEVEL_CHOICES!r}: 'grna' fits y ~ fraction:grna + rowID + "
f"columnID, 'gene' fits y ~ gene_fraction:gene + rowID + "
f"columnID, and 'both' fits each of them SEPARATELY.")
if regression_type == 'mixed':
return ('gene',)
if key == 'both':
return ('grna', 'gene')
return (key,)
def _wide_fixed_effect_design(df, dependent_variable, *, level,
model_plate_position=True,
block_screen=False, intercept='fitted'):
"""Build one fixed-effects row per independent well from long fractions.
This is the explicit long-to-wide alternative to Patsy's historical
long-row interaction formula. Each guide/gene becomes a numeric fraction
column, absent predictors are zero, and the response and metadata must be
constant within a well. The returned feature names keep spaCR's existing
``fraction:grna[...]`` / ``gene_fraction:gene[...]`` contract so every
estimator and results consumer can use the same coefficient path.
"""
from .regression_layout import long_to_wide_regression_data
from .schema import SCREEN_KEY
if level == 'grna':
predictor, value, prefix = 'grna', 'fraction', 'fraction:grna['
elif level == 'gene':
predictor, value, prefix = 'gene', 'gene_fraction', 'gene_fraction:gene['
else:
raise ValueError(f"wide model design needs one level, got {level!r}")
metadata = [dependent_variable, 'plateID', 'rowID', 'columnID']
if 'cell_count' in df.columns:
metadata.append('cell_count')
if block_screen and SCREEN_KEY in df.columns:
metadata.append(SCREEN_KEY)
wide = long_to_wide_regression_data(
df,
index_columns='prc',
predictor_column=predictor,
value_column=value,
metadata_columns=metadata,
fill_value=0.0,
).sort_values('prc', kind='stable').reset_index(drop=True)
identifiers = sorted(
set(map(str, df[predictor].dropna().astype(str).unique()))
)
missing = [name for name in identifiers if name not in wide.columns]
if missing:
raise AssertionError(
"wide pivot lost predictor columns: " + ", ".join(missing[:5])
)
predictors = wide[identifiers].apply(pd.to_numeric, errors='raise').copy()
predictors.columns = [prefix + name + ']' for name in identifiers]
design_parts = []
mode = str(intercept or 'fitted').strip().lower()
if mode not in INTERCEPT_MODES:
raise ValueError(
f"intercept={intercept!r} is not one of {list(INTERCEPT_MODES)}"
)
if mode in ('fitted', 'control'):
design_parts.append(pd.DataFrame({'Intercept': np.ones(len(wide))}))
design_parts.append(predictors.reset_index(drop=True))
nuisance_columns = []
if model_plate_position:
nuisance_columns.extend(['plateID', 'rowID', 'columnID'])
if block_screen:
nuisance_columns.append(SCREEN_KEY)
for column in nuisance_columns:
if column not in wide.columns:
raise ValueError(
f"model_data_layout='wide' needs nuisance column {column!r}"
)
values = wide[column].astype(str)
levels = sorted(values.unique())
for category in levels[1:]:
design_parts.append(pd.DataFrame({
f'{column}[T.{category}]': values.eq(category).astype(float)
}))
X = pd.concat(design_parts, axis=1)
y = pd.DataFrame({
dependent_variable: pd.to_numeric(
wide[dependent_variable], errors='raise'
).to_numpy(dtype=float)
})
return y, X, wide
[docs]
def regression(df, csv_path, dependent_variable='predictions', regression_type=None, alpha=1.0,
random_row_column_effects=False, nc='233460', pc='220950', controls=None,
dst=None, cov_type=None, plot=False, l1_ratio=0.5, quantile=0.5,
hinge_threshold=None, hinge_n_boot=200, huber_t=1.345, qc=True,
spline_knots=4, spline_degree=3,
legacy_volcano=False, level='grna', level_dst=None,
draw_shared_panels=True, group_lasso_lambda='auto',
rra_alpha=0.25, rra_permutations=10000,
model_plate_position=True,
regression_backend=DEFAULT_REGRESSION_BACKEND,
verbose=False, transform="",
intercept='fitted', intercept_value=0.0,
model_data_layout='long'):
"""Run the full regression pipeline: clean, fit, extract coefficients, optional volcano plot.
:param df: Long-format DataFrame with gRNA/gene fractions and the
dependent variable.
:param csv_path: Path used to derive the volcano-plot filename.
:param dependent_variable: Response column name. Default
``'predictions'``.
:param regression_type: Model type; auto-selected via
:func:`check_distribution` when ``None``.
:param regression_backend: WHO fits it, one of
:data:`REGRESSION_BACKEND_ORDER`. Default ``'statsmodels'``. It is
threaded to whichever fitter this run reaches -- the mixed branch and
:func:`regression_model` alike -- so one setting answers for the
whole run, and a backend that cannot fit the chosen family is refused
by name before any design is built.
:param alpha: Regularisation strength for penalised models.
:param random_row_column_effects: If True, fit a mixed model with
random row/column effects.
:param model_plate_position: Whether ``plateID``, ``rowID`` and
``columnID`` are terms in a fixed-effects model (or the corresponding
grouping/variance structure in a mixed model). Direct calls default
to ``True`` for API
compatibility; new application settings default to ``False`` so the
terms are opt-in. See :func:`prepare_formula` for the measured costs
of including or omitting them. ``False`` with
``random_row_column_effects=True`` is refused: there is nothing left
to make random.
:param nc: Negative-control gene identifier. Default ``'233460'``.
:param pc: Positive-control gene identifier. Default ``'220950'``.
:param controls: Explicit list of control identifiers.
:param dst: Output directory for plots and summaries.
:param cov_type: Optional covariance estimator for the likelihood fits.
:param plot: If True, render the volcano plot after fitting.
:param l1_ratio: ``elasticnet`` L1/L2 mix.
:param quantile: Quantile fitted by ``quantile`` regression.
:param hinge_threshold: Response cut used to binarise for ``hinge``.
:param hinge_n_boot: Bootstrap resamples behind the hinge p-values.
:param huber_t: Huber tuning constant for ``rlm``/``huber``.
:param group_lasso_lambda: Block penalty for ``group_lasso``.
:param rra_alpha: Top fraction of the guide ranking ``rra`` aggregates.
:param rra_permutations: Draws per guide count in ``rra``'s null.
:param qc: Write the regression QC suite into ``<dst>/regression_qc/``.
:param legacy_volcano: also draw the ORIGINAL matplotlib volcano.
Default ``False``. The interactive one is far faster and the
house-style panel is what a run now produces; drawing both gives two
volcanoes in two idioms on the same grid.
Requires ``dst`` and a design matrix, so it is skipped for the mixed
branch and when no destination was given.
:param level: WHICH MODEL TO FIT -- ``'grna'`` (default) or ``'gene'``.
One level, one design; ``'both'`` is refused here because it is two
fits. :func:`regression_levels` is the entry point that does both.
Ignored by ``regression_type='mixed'``, which fits the gene fixed and
the guide random inside it and so is already both levels.
:param level_dst: Where THIS LEVEL's figures go -- the QC suite, the
volcano, the publication sheet. Defaults to ``dst``, which is what a
single-level run wants. :func:`regression_levels` gives each level its
own subfolder so two fits cannot overwrite each other's
``regression_figure.pdf``.
:param draw_shared_panels: Draw the guide-fraction and response
distributions, which describe the DATA and not the fit. False on the
second of two fits, so the figure grid gets one copy rather than two
identical ones.
:param model_data_layout: ``'long'`` preserves the historical formula
with one fitted row per well-guide pair. ``'wide'`` pivots fractions
to one row per independent well before any fixed-effects estimator is
fitted. Mixed models require their long nesting representation and
therefore use long data even when a wide count input was supplied.
:returns: ``(model, coef_df, regression_type)``.
"""
if controls is None:
controls = ['']
from .plot import volcano_plot, plot_histogram
level_dst = dst if level_dst is None else level_dst
volcano_path = create_volcano_filename(
csv_path, regression_type,
quantile if regression_type == 'quantile' else alpha, level_dst)
if regression_type is None:
regression_type = check_distribution(df[dependent_variable])
wanted = resolve_levels(regression_type, level)
if len(wanted) != 1:
raise ValueError(
f"regression() fits ONE level; level={level!r} asks for "
f"{list(wanted)}. Call regression_levels(), which fits each of "
f"them separately and corrects each within itself.")
level = wanted[0]
model_layout = str(model_data_layout or 'long').strip().lower()
if model_layout not in {'long', 'wide'}:
raise ValueError(
f"model_data_layout={model_data_layout!r}; choose 'long' or 'wide'."
)
print(f"Using regression type: {regression_type}")
dependent_variable, transform, glm_force_identity, conflict_note = (
resolve_glm_transform_conflict(
dependent_variable, transform=transform,
available=getattr(df, 'columns', ()),
regression_type=regression_type))
if conflict_note:
print(conflict_note)
df = check_and_clean_data(df, dependent_variable)
intercept_offset = 0.0
intercept_mode = str(intercept or 'fitted').strip().lower()
if intercept_mode == 'control':
df, intercept_offset = centre_on_controls(df, dependent_variable, nc)
if intercept_offset:
print(f"Intercept set to the negative controls: {dependent_variable} "
f"centred by {intercept_offset:.6g}, so a coefficient reads "
f"as its distance from {nc!r}.")
else:
print(f"Intercept left as fitted: no rows match "
f"negative_control_id={nc!r}, so there is no control level to "
f"centre on.")
elif intercept_mode == 'value':
intercept_offset = float(intercept_value or 0.0)
if intercept_offset:
df = df.copy()
df[dependent_variable] = (
np.asarray(df[dependent_variable], dtype=float)
- intercept_offset)
print(f"Intercept pinned at {intercept_offset:.6g}: every "
f"coefficient reads as its distance from that value.")
qc_design = None
fit_frame = None
fit_counts = {}
block_screen = screen_is_blockable(df)
if block_screen:
print(f"Blocking on {df['screenID'].nunique()} screens: "
f"{sorted(df['screenID'].astype(str).unique())}")
if regression_type == 'mixed' or random_row_column_effects:
if model_layout == 'wide':
print("model_data_layout='wide' was normalized back to long for "
"the mixed model because guide-within-gene nesting is a "
"long-data structure.")
regression_type = 'mixed'
level = 'gene'
formula = prepare_formula(
dependent_variable,
random_row_column_effects=random_row_column_effects,
block_screen=block_screen, level='gene',
model_plate_position=model_plate_position,
intercept=intercept)
mixed_model, coef_df = fit_mixed_model(
df, formula, level_dst,
random_row_column_effects=random_row_column_effects,
regression_backend=regression_backend)
model = mixed_model
observed_count = getattr(model, 'nobs', getattr(model, 'n_obs', None))
if observed_count is not None and np.isfinite(observed_count):
fit_counts['n_rows_fitted'] = int(observed_count)
inner = getattr(model, 'model', None)
row_labels = getattr(getattr(inner, 'data', None), 'row_labels', None)
if row_labels is not None and df.index.is_unique:
fit_frame = df.loc[row_labels]
elif fit_counts.get('n_rows_fitted') == len(df):
fit_frame = df
exog = getattr(inner, 'exog', None)
if exog is not None:
fit_counts['n_design_columns'] = int(exog.shape[1])
elif getattr(model, 'k_fe', None) is not None:
fit_counts['n_design_columns'] = int(model.k_fe)
fit_counts['layout'] = 'long'
else:
formula = prepare_formula(dependent_variable,
random_row_column_effects=False,
block_screen=block_screen, level=level,
model_plate_position=model_plate_position,
intercept=intercept)
fit_df = df
if model_layout == 'wide':
y, X, fit_df = _wide_fixed_effect_design(
df, dependent_variable, level=level,
model_plate_position=model_plate_position,
block_screen=block_screen, intercept=intercept,
)
print(f"Model data pivoted to {len(fit_df)} independent well "
f"rows and {X.shape[1]} design columns.")
else:
y, X = dmatrices(formula, data=df, return_type='dataframe')
model_index = y.index
if model_index.equals(fit_df.index):
fit_frame = fit_df
elif fit_df.index.is_unique:
fit_frame = fit_df.loc[model_index]
fit_counts = {'n_rows_fitted': int(len(y)),
'n_design_columns': int(X.shape[1]),
'layout': model_layout}
if draw_shared_panels and not _show_well_distributions(
df, dependent_variable, dst, plot=plot):
plot_histogram(y, dependent_variable, dst=dst)
plot_histogram(df, 'fraction', dst=dst)
print('Data will not be scaled: the design is fractions and dummies '
'on one common scale, and scaling it per column would rescale '
'each gRNA coefficient by a different constant.')
weights = (fit_df['cell_count'].loc[model_index]
if 'cell_count' in fit_df.columns else None)
groups = None
print(f'Performing {regression_type} {level}-level regression')
model = regression_model(
X, y,
regression_type=regression_type,
groups=groups,
alpha=alpha,
cov_type=cov_type,
weights=weights,
l1_ratio=l1_ratio,
quantile=quantile,
hinge_threshold=hinge_threshold,
huber_t=huber_t,
spline_knots=spline_knots,
spline_degree=spline_degree,
exposure=weights,
group_lasso_lambda=group_lasso_lambda,
rra_alpha=rra_alpha,
rra_permutations=rra_permutations,
regression_backend=regression_backend,
verbose=verbose,
response_name=str(y.name) if hasattr(y, 'name') else '',
transform=transform,
glm_force_identity=glm_force_identity,
)
fitted_exog = getattr(getattr(model, 'model', None), 'exog', None)
if fitted_exog is not None:
fit_counts['n_design_columns'] = int(fitted_exog.shape[1])
coef_df = process_model_coefficients(
model, regression_type, X, y, nc, pc, controls,
hinge_threshold=hinge_threshold, hinge_n_boot=hinge_n_boot)
display(coef_df)
qc_design = (X, y)
if fit_frame is not None:
contributing = fit_frame
if fit_counts.get('layout') == 'wide' and 'prc' in fit_frame:
contributing = df.loc[df['prc'].isin(fit_frame['prc'])]
for name, column in (('n_wells', 'prc'), ('n_guides', 'grna'),
('n_genes', 'gene')):
fit_counts[name] = int(contributing[column].nunique())
if plot and legacy_volcano:
volcano_plot(
coef_df,
fold_change_col='coefficient',
p_value_col='p_value',
name_col='feature',
x_transform='none',
save_path=volcano_path,
show=False,
)
qc_manifest = None
if qc and qc_design is not None and level_dst:
qc_manifest = _write_regression_qc(
model, qc_design[0], qc_design[1], fit_df, level_dst,
coef_df=coef_df, regression_type=regression_type,
volcano_path=volcano_path if plot else None)
if level_dst:
_show_house_style_panels(coef_df, plot=plot)
_write_regression_sheet(coef_df, level_dst)
try:
from .figures.summary import summarise
text = summarise(coef_df)
if text:
import textwrap
print()
print("SUMMARY")
print(textwrap.fill(text, 88))
except Exception as error: # noqa: BLE001 - never lose a run over prose
print(f"Could not summarise the run: {error}")
coef_df = coef_df.copy()
coef_df['level'] = level
coef_df.attrs['fit_design'] = fit_counts
if qc_manifest is not None and coef_df is not None:
coef_df.attrs["qc_manifest"] = qc_manifest
return model, coef_df, regression_type
[docs]
def regression_levels(df, csv_path, dependent_variable='predictions',
regression_type=None, level='both', dst=None, **kwargs):
"""Fit every level the run asked for, SEPARATELY, and return one per level.
THIS IS THE TWO-FIT ENTRY POINT, and the reason it exists is that the one
design spaCR used to fit cannot be fitted at all. ``gene_fraction`` is the
SUM of the gene's gRNA fractions, so
``y ~ fraction:grna + gene_fraction:gene + plateID + rowID + columnID``
puts a block of columns and their own sums into one design. Measured on
the reference TSG101 screen: 1248 parameters at rank 862 -- a
386-dimensional EXACT null space -- and the fit statsmodels returned had a
residual sum of squares bit-identical to the one you get by adding seven
times a null vector to it. See :data:`COLLINEAR_FORMULA_FRAGMENT`.
Two fits, two tables, TWO CORRECTIONS. Each fit is its own
multiple-testing family and is corrected within itself. Pooling them would
be wrong twice over: they are not independent -- same wells, and the gene
regressor IS the sum of the guide regressors -- and doubling the family
size costs power for no protection. :func:`perform_regression` applies the
correction per level and writes ``results_grna.csv`` and
``results_gene.csv``.
``regression_type='mixed'`` fits ONCE and returns one entry, ``'gene'``:
that model has both levels inside it already, the gene as a fixed effect
and the guide as a random effect nested in the gene. Its guide output is
BLUPs, which is why it cannot be split into two testing families.
:param df: long-format DataFrame of gRNA/gene fractions and the
dependent variable, passed to :func:`regression` for every level.
:param csv_path: path passed to :func:`regression`, which derives the
volcano-plot filename from it.
:param level: ``'both'`` (default), ``'grna'`` or ``'gene'``.
:param dst: the run folder. With more than one fit each level's FIGURES go
into ``<dst>/<level>/`` so they cannot overwrite each other; the
tables stay in ``<dst>``, where every consumer looks for them.
:param kwargs: passed straight through to :func:`regression`.
:returns: ``dict`` mapping level to ``(model, coef_df, regression_type)``,
in fit order.
:raises ValueError: for a level that is not one of :data:`LEVEL_CHOICES`.
"""
import os
levels = resolve_levels(regression_type, level)
if regression_type == 'mixed' and str(level).strip().lower() != 'both':
print(f"regression_type='mixed' fits the gene fixed with guides "
f"random nested inside, so it is already both levels and "
f"level={level!r} is not read. Its guide output is BLUPs, not "
f"coefficients with p-values.")
fits = {}
for index, one in enumerate(levels):
level_dst = dst
if dst and len(levels) > 1:
level_dst = os.path.join(str(dst), one)
os.makedirs(level_dst, exist_ok=True)
print(f"Fitting level {index + 1} of {len(levels)}: {one}")
fits[one] = regression(
df, csv_path, dependent_variable=dependent_variable,
regression_type=regression_type, dst=dst, level=one,
level_dst=level_dst, draw_shared_panels=(index == 0), **kwargs)
regression_type = fits[one][2]
return fits
def _show_well_distributions(frame, response_name, dst, plot=True):
"""Draw the guide-fraction and response distributions in the house style.
:returns: True when they were drawn. False sends the caller back to the
original ``plot_histogram``, because a figure is not worth losing a
fit over.
"""
try:
import matplotlib.pyplot as plt
from .figures import distributions
except Exception as error: # noqa: BLE001
print(f"Could not load the distribution panels: {error}")
return False
drawn = 0
per_panel = {"response": {"column": response_name}, "guide_fraction": {}}
for key in distributions.ORDER:
try:
figure, panel = distributions.build_panel(
key, frame, **per_panel.get(key, {}))
except Exception as error: # noqa: BLE001
print(f"Distribution panel {key} did not draw: {error}")
continue
if not getattr(panel, "drawn", False):
plt.close(figure)
continue
figure.set_label(panel.title)
figure._spacr_title = panel.title
if dst:
try:
from .plot import save_figure
name = distributions.FILENAMES[key].format(
response=response_name)
save_figure(figure, os.path.join(str(dst), f"{name}.pdf"),
bbox_inches="tight")
except Exception:
pass
if plot:
plt.show()
plt.close(figure)
drawn += 1
return drawn > 0
def _show_plates(frame, variable, dst):
"""Draw every plate as one small multiple. True when it was drawn."""
try:
import matplotlib.pyplot as plt
from .figures.plates import build_plates
except Exception as error: # noqa: BLE001
print(f"Could not load the plate panel: {error}")
return False
try:
figure, panel = build_plates(frame, variable, grouping="mean",
min_max="allq", min_count=0)
except Exception as error: # noqa: BLE001
print(f"The plate panel did not draw: {error}")
return False
if not getattr(panel, "drawn", False):
plt.close(figure)
return False
figure.set_label(panel.title)
figure._spacr_title = panel.title
if dst:
try:
from .plot import save_figure
save_figure(
figure,
os.path.join(str(dst), f"plate_heatmap_{variable}.pdf"),
bbox_inches="tight")
except Exception:
pass
plt.show()
plt.close(figure)
return True
def _show_house_style_panels(coef_df, plot=True):
"""Draw each house-style panel and hand it to whatever is watching.
One figure per panel rather than one sheet, because the grid puts each on
its own lettered cell and a single composite would be one unreadable
tile. The sheet is written to disk as well, for the version that goes in
a paper.
Never fatal: the fit is already done and losing a run over a figure would
be the worst possible trade.
"""
if coef_df is None or not len(coef_df):
return 0
try:
import matplotlib.pyplot as plt
from .figures import SHEET_ORDER, build_panel
except Exception as error: # noqa: BLE001
print(f"Could not load the figure style: {error}")
return 0
shown = 0
for key in SHEET_ORDER:
try:
figure, panel = build_panel(key, coef_df)
except Exception as error: # noqa: BLE001
print(f"Panel {key} did not draw: {error}")
continue
if not panel.drawn:
plt.close(figure)
continue
figure.set_label(panel.title)
figure._spacr_title = panel.title
if plot:
plt.show()
plt.close(figure)
shown += 1
if shown:
print(f"Drew {shown} regression panels in the house style.")
return shown
def _write_regression_sheet(coef_df, dst):
"""Write ``<dst>/regression_figure.pdf`` and its legend.
Never fatal: a fit that produced a coefficient table has already done the
work, and losing the run because a panel could not be drawn would be the
worst possible trade.
"""
import os
if coef_df is None or not len(coef_df):
return None
try:
from .figures import build_sheet
sheet = build_sheet(coef_df, width='double', target='print')
folder = str(dst)
os.makedirs(folder, exist_ok=True)
path = os.path.join(folder, 'regression_figure.pdf')
from .figure_sink import publish
path = publish(sheet.figure, path, bbox_inches='tight') or path
with open(os.path.join(folder, 'regression_figure_legend.txt'),
'w') as handle:
handle.write(sheet.legend() + '\n')
try:
import matplotlib.pyplot as plt
plt.close(sheet.figure)
except Exception:
pass
print(f"Wrote the regression figure to {path} "
f"({len(sheet.panels)} panels"
+ (f", {len(sheet.skipped)} not applicable"
if sheet.skipped else '') + ').')
return path
except Exception as error: # noqa: BLE001 - never lose a run over a figure
print(f"Could not draw the regression figure: {error}")
return None
#: What a run's statsmodels summary is written as, and every older name the
#: reader still accepts. NEWEST FIRST -- the first one found wins.
#:
#: The name used to be ``mode_summary.csv``, which was wrong twice over:
#: "mode" is a typo for "model", and the content is the statsmodels TEXT
#: summary, never CSV. A name that does not follow the file is a path nobody
#: can open, so the format is corrected going forward and the old names are
#: still READ -- a run finished last month keeps its summary.
#: How many coefficient rows the console will print before it stops and
#: points at the file instead. The header of a statsmodels summary is about
#: twenty lines; a screen's coefficient table is hundreds.
CONSOLE_COEFFICIENT_LIMIT = 12
[docs]
def fit_quality_note(model) -> str:
"""Return a one-line goodness-of-fit summary for a fitted GLM.
McFadden's pseudo-R-squared compares log-likelihoods, and it is the
appropriate summary for a GLM with a discrete response. A Gaussian
identity-link fit instead reports ordinary R-squared because its
likelihood is a density and the McFadden ratio is not interpretable on
the usual zero-to-one scale.
:param model: a fitted statsmodels GLM result.
:returns: A labelled goodness-of-fit line for the console.
"""
family = getattr(model, 'family', None)
if isinstance(family, sm.families.Gaussian):
try:
resid = np.asarray(model.resid_response, dtype=float).reshape(-1)
observed = resid + np.asarray(
model.fittedvalues, dtype=float).reshape(-1)
centred = observed - observed.mean()
total = float(np.dot(centred, centred))
residual = float(np.dot(resid, resid))
except (AttributeError, TypeError, ValueError):
return "R²: not available for this fit"
if not np.isfinite(total) or total <= 0:
return "R²: not available for this fit (the response is constant)"
return (f"R²: {1.0 - residual / total:.4f} (ordinary R², not "
f"McFadden -- this is a Gaussian identity-link fit)")
try:
null_value = model.llnull
if null_value is None:
raise AttributeError("no null log-likelihood on this result")
llf, null = float(model.llf), float(null_value)
except (AttributeError, TypeError, ValueError):
try:
llf = float(model.llf)
null = float(model.null_deviance) / -2.0
except (AttributeError, TypeError, ValueError):
return "McFadden's R²: not available for this fit"
if not np.isfinite(null) or null == 0:
return "McFadden's R²: not available for this fit"
return mcfadden_note(1.0 - (llf / null))
[docs]
def mcfadden_note(r2) -> str:
"""Format McFadden's pseudo-R² and flag a negative value.
A negative value means the fitted model predicts the response worse than
an intercept-only model. The returned note explains that the coefficients
should not be interpreted and points to a common cause: applying a response
transform that duplicates the fitted family's link.
:param r2: Pseudo-R² value, or a value convertible to ``float``.
:returns: One-line diagnostic text suitable for a console or report.
"""
try:
value = float(r2)
except (TypeError, ValueError):
return "McFadden's R²: not available for this fit"
if value < 0:
return (
f"McFadden's R²: {value:.4f} <-- NEGATIVE. This fit predicts the "
f"response WORSE than its own intercept, so its coefficients and "
f"P values do not describe the data. The usual cause is a "
f"response transformed twice: check that `transform` is not "
f"applying a log or logit that the family's link already applies."
)
return f"McFadden's R²: {value:.4f}"
[docs]
def summary_for_console(model, *, verbose=False,
limit=CONSOLE_COEFFICIENT_LIMIT) -> str:
"""Return a statsmodels summary sized for terminal output.
When the coefficient table exceeds ``limit``, the diagnostic header and
notes are retained while the table is replaced by a pointer to the saved
summary and the sortable Coefficients view. Set ``verbose=True`` to return
the complete statsmodels rendering.
:param model: Fitted model result with a ``summary()`` method.
:param verbose: Return the complete summary regardless of its size.
:param limit: Maximum coefficient rows printed in compact mode.
:returns: Complete or compact plain-text model summary.
"""
try:
text = str(model.summary())
except Exception as error: # noqa: BLE001
return (f"statsmodels could not render a summary for this fit "
f"({type(error).__name__}: {error}).")
if verbose:
return text
lines = text.splitlines()
header = next((i for i, line in enumerate(lines)
if "coef" in line and "std err" in line), None)
if header is None:
return text
rule = next((i for i in range(header + 1, len(lines))
if set(lines[i].strip()) == {"-"}), None)
if rule is None:
return text
end = next((i for i in range(rule + 1, len(lines))
if set(lines[i].strip()) == {"="}), len(lines))
rows = [line for line in lines[rule + 1:end] if line.strip()]
if len(rows) <= limit:
return text
return "\n".join(
lines[:rule + 1]
+ [f" {len(rows)} coefficients — not printed here. They are in the "
f"run's model_summary.txt and in the Coefficients tab, which sorts "
f"and filters them. Set verbose=True to print them."]
+ lines[end:])
SUMMARY_FILENAME = 'model_summary.txt'
SUMMARY_FILENAMES = (SUMMARY_FILENAME, 'mode_summary.csv', 'summary.csv')
[docs]
def save_summary_to_file(model, file_path=SUMMARY_FILENAME):
"""
Write ``model.summary().as_text()`` to ``file_path`` as plain text.
The content is the statsmodels text summary, never CSV -- which is why
the default name is :data:`SUMMARY_FILENAME` and no longer
``summary.csv``. Older runs on disk wrote ``mode_summary.csv``; every
reader in this repository accepts both, see :data:`SUMMARY_FILENAMES`.
:param model: Fitted statsmodels results object.
:param file_path: Destination path. Default :data:`SUMMARY_FILENAME`.
:returns: the path written, or ``None`` if there was nothing to write.
NEVER RAISES INTO A FINISHED RUN. This is called after every table has
been written; a backend whose ``summary()`` throws must not take the run
down with it, and the caller is told by the ``None`` rather than by a
traceback.
"""
summary = getattr(model, 'summary', None)
if not callable(summary):
return None
try:
summary_str = summary().as_text()
except Exception as error: # noqa: BLE001 - a summary is not worth a run
print(f"Could not render the model summary: "
f"{type(error).__name__}: {error}")
return None
folder = os.path.dirname(os.path.abspath(file_path))
os.makedirs(folder, exist_ok=True)
with open(file_path, 'w') as f:
f.write(summary_str)
return file_path
def _split_prc(text):
"""Return ``(plateID, rowID, columnID)`` for one ``prc`` well key.
Parse from right to left because only the leading plate ID may contain the
key separator. This preserves plate names such as ``'exp1_plate1'``.
The row and column are returned exactly as they appear — nothing is
canonicalised, because the caller rebuilds ``prc`` from these columns and
a rewritten token would change the identity rows are joined on.
Unescape the plate component to match :func:`spacr.schema.compose_prc` and
:func:`spacr.schema.parse_prcf`; return row and column tokens unchanged.
For keys with more than three components, require the final tokens to be
a recognizable row/column pair. This accepts a plate containing the
separator while rejecting a deeper ``prcf`` or ``prcfo`` key. Exactly
three components remain accepted without positional-token validation.
:param text: a ``prc`` key, e.g. ``'plate1_r1_c1'``.
:returns: ``(plateID, rowID, columnID)``.
:raises spacr.schema.KeyParseError: when ``text`` has fewer than three
components, i.e. it is not a well key at all, or when it has more
than three and the trailing pair is not a row and a column.
"""
key = str(text).strip()
parts = key.split(schema.KEY_SEPARATOR)
if len(parts) < 3:
raise schema.KeyParseError(
f'{text!r} is not a prc: expected plate_row_column, got '
f'{len(parts)} component(s).')
plate = schema.KEY_SEPARATOR.join(parts[:-2])
row, column = parts[-2], parts[-1]
if not plate.strip():
raise schema.KeyParseError(
f'{text!r} is not a prc: it has no plate.')
if not row.strip() or not column.strip():
raise schema.KeyParseError(
f'{text!r} is not a prc: its row is {row!r} and its column is '
f'{column!r}, and an empty one identifies no well — every well of '
f'{plate!r} would be grouped together.')
if len(parts) > 3 and not _is_row_column_pair(row, column):
raise schema.KeyParseError(
f'{text!r} is not a prc: it has {len(parts)} components and its '
f'last two, {row!r} and {column!r}, are not a row and a column. '
f'{_name_deeper_key(parts)}'
f'If this really is a plate id containing '
f'{schema.KEY_SEPARATOR!r}, its row and column must be written '
f'the way spaCR writes them (r<N>/letters and c<N>/digits) for '
f'the plate to be separable from them.')
return schema.unescape_filename_component(plate), row, column
#: ``prc`` for a whole frame. :mod:`spacr.schema` owns it -- one place
#: composes a key -- and this name is kept because seven call sites in this
#: module use it.
_compose_prc_column = schema.compose_prc_column
def _is_row_column_pair(row, column):
"""True when ``(row, column)`` is recognisably a well's row and column.
Deliberately narrow: it is the guard that stops :func:`_split_prc` from
absorbing a ``prcf`` into an underscored plate id, so it must reject a
``(columnID, fieldID)`` pair and a ``(fieldID, objectID)`` pair.
:param row: candidate ``rowID`` token.
:param column: candidate ``columnID`` token.
:returns: whether the pair can be a row and a column.
"""
row_text, column_text = str(row).strip(), str(column).strip()
if not row_text or not column_text:
return False
if schema.is_positional_pair(row_text, column_text):
return True
if row_text[:1].lower() == schema.KEY_PREFIXES[schema.ROW_KEY]:
row_ok = schema.row_index(row_text) is not None
else:
row_ok = schema.row_index_from_letters(row_text) is not None
if not row_ok:
return False
if column_text[:1].lower() == schema.KEY_PREFIXES[schema.COLUMN_KEY]:
return schema.column_index(column_text) is not None
return column_text.isdigit()
def _name_deeper_key(parts):
"""Return a sentence naming the deeper key ``parts`` looks like, or ''.
Split out of :func:`_split_prc` only so the error it raises can say
*which* mistake was made instead of describing the shape and leaving the
caller to work it out.
:param parts: the separator-split components of the rejected key.
:returns: a sentence ending in a space, or ``''`` when the key does not
look like a ``prcf`` / ``prcfo`` / timepoint key.
"""
tail = parts[-1]
if schema.object_index(tail) is not None and len(parts) >= 5:
return ('That is a prcfo (plate_row_column_field_object); '
'_split_prc takes a prc. Use schema.parse_prcfo. ')
if schema.field_index(tail) is not None:
return ('That is a prcf (plate_row_column_field); _split_prc takes a '
'prc. Use schema.parse_prcf, or drop the field first. ')
if schema.time_index(tail) is not None:
return ('That ends in a timepoint; a prc has none. Aggregate the '
'timepoints away before keying on the well. ')
return ''
def _qc_graph_type(fallback: str = 'jitter_bar') -> str:
"""The graph type the regression QC figures should start on.
The DEFAULT GRAPH TYPE setting decides what is drawn FIRST, for every
graph in spaCR and not only for Regression. The three QC figures below
hardcoded ``'jitter_bar'`` and ignored it.
THE FALLBACK IS THE OLD LITERAL, deliberately. :func:`graph_types.
start_for` answers the CALLER'S OWN starting form when the user has
expressed no preference, so a user who has set nothing sees exactly the
figure they saw before. A preference nobody expressed must not move an
existing view -- which is the rule ``fast_plots`` already follows for
``DEFAULT_MARK``.
THE TWO VOCABULARIES ARE NOT THE SAME, and this is where they meet.
``graph_types`` stores ``'bar_jitter'``; :class:`spacr.plot.spacrGraph`
draws ``'jitter_bar'``. :func:`graph_types.mark_for` is the translation,
and skipping it hands matplotlib a name that its own error message lists
as unknown. Every type that FITS ``categorical_continuous`` translates to
something ``spacrGraph`` draws -- checked, all six.
The shape is ``categorical_continuous`` because all three figures are one
measurement grouped by ``plateID``.
NO NOTE, BECAUSE NO COUNTS. :func:`graph_types.start_for` explains a
swapped graph only when it is handed the per-group sizes, and these
figures are drawn from a CSV path, not from sizes the caller holds. A note
computed without counts is always empty, so the print that used to follow
this call could never run. Passing counts is not the small fix it looks
like: the fallback here is a MARK spelling, and on the too-thin path
`start_for` re-checks it with `fits`, which a mark spelling fails -- see
:func:`graph_types.mark_to_start_on`, which records exactly that. The swap
concerns bar, box, violin and line drawn over 8 or fewer observations in a
group, and these figures group whole plates of wells.
:param fallback: what to draw when no preference is stored. The default
is the literal these figures used before this existed.
:returns: the graph type in ``spacrGraph``'s vocabulary.
"""
from .graph_types import mark_to_start_on
return mark_to_start_on('categorical_continuous', fallback)[0]
def _assign_prc_parts(df, column=schema.PRC_KEY,
columns=schema.WELL_KEY_COLUMNS):
"""Split ``df[column]`` into plate / row / column and assign them onto ``df``.
The frame-level counterpart of :func:`_split_prc`, and the ``prc`` sibling
of :func:`_assign_prcfo_parts`.
:param df: frame carrying ``column``.
:param column: name of the ``prc`` column. Default ``'prc'``.
:param columns: names to assign, in plate / row / column order.
:returns: ``df``, mutated in place and returned for chaining.
:raises spacr.schema.KeyParseError: when any value is not a ``prc``.
"""
parsed = [_split_prc(value) for value in df[column]]
for position, name in enumerate(columns):
df[name] = [part[position] for part in parsed]
return df
[docs]
def resolve_auto_inference(data, settings, *, well_column='prc',
guide_column='grna'):
"""Choose ``analysis_mode`` for ``inference='auto'`` from the design.
The simultaneous model estimates one coefficient per guide from the wells,
so it needs more wells than guides -- with an intercept and any plate fixed
effects on top -- before those coefficients are identifiable at all. Below
that the design matrix is rank deficient: statsmodels still returns a
number for every guide, but the numbers are one arbitrary solution out of
infinitely many, and their P values describe nothing.
That is not a hypothetical. The screen this was written for has 824 guides
in 587 analysed wells; the published fit had 825 parameters, rank 579 and
8 residual degrees of freedom, and refitting it did not reproduce its own
coefficients.
``auto`` therefore picks the permutation test whenever the simultaneous fit
would be unidentifiable, and says so. It is deliberately conservative: it
needs a real margin (``_IDENTIFIABILITY_MARGIN`` wells per guide) rather
than a bare majority, because a design that only just fits is one dropped
well away from not fitting.
Anything other than ``inference='auto'`` is returned untouched, so an
explicit choice is never overridden.
:param data: the analysis table (a DataFrame); its distinct well and
guide counts, and the permutation block column when present, size
the design.
:param settings: run settings; ``inference``, ``analysis_mode``,
``analysis_unit``, ``agg_type`` and ``guide_permutation_block`` are
read. It is not modified.
:param well_column: column whose distinct values count the wells.
:param guide_column: column whose distinct values count the guides.
:returns: ``(analysis_mode, reason)``. ``reason`` is a sentence naming the
counts, suitable for the log and for the Methods section.
"""
inference = str(settings.get('inference', 'auto')).strip().lower()
if inference != 'auto':
return settings.get('analysis_mode', 'regression'), (
f"inference={inference!r} was set explicitly.")
per_object = str(settings.get('analysis_unit') or 'well').lower() != 'well'
aggregated = settings['agg_type'] if 'agg_type' in settings else 'mean'
if per_object or aggregated is None:
return 'regression', (
"auto chose the simultaneous model: the rows are one per OBJECT "
"(agg_type is None or analysis_unit is not 'well'), and the "
"permutation test needs one row per well. Set an agg_type such "
"as 'mean' if the permutation test is wanted.")
try:
n_wells = int(data[well_column].nunique())
n_guides = int(data[guide_column].nunique())
except (KeyError, TypeError):
return 'guide_permutation', (
"The design could not be measured, so the permutation test was "
"used because it is valid regardless of the number of guides.")
blocks = 0
block_column = str(settings.get('guide_permutation_block', 'plateID'))
if block_column in getattr(data, 'columns', ()):
blocks = max(int(data[block_column].nunique()) - 1, 0)
parameters = 1 + blocks + n_guides
required = parameters * _IDENTIFIABILITY_MARGIN
if n_wells >= required:
return 'regression', (
f"auto chose the simultaneous model: {n_wells} analysed wells for "
f"{parameters} parameters ({n_guides} guides + intercept + "
f"{blocks} block terms), at least the {_IDENTIFIABILITY_MARGIN}x "
f"margin required.")
return 'guide_permutation', (
f"auto chose the permutation test: {n_wells} analysed "
f"wells cannot identify {parameters} simultaneous parameters "
f"({n_guides} guides + intercept + {blocks} block terms). Each guide "
f"is tested as a marginal association instead. Set "
f"inference='parametric' to force the simultaneous fit.")
#: How many wells per estimated parameter ``auto`` insists on before it will
#: choose the simultaneous model. 1.0 would accept a design with zero residual
#: degrees of freedom, which fits perfectly and tests nothing.
_IDENTIFIABILITY_MARGIN = 2.0
#: Fraction of count wells that must survive the score join before the run is
#: allowed to continue. Below this the two inputs are describing different
#: plates, and every number downstream is computed on whatever happened to
#: overlap.
_MINIMUM_PAIRED_WELL_FRACTION = 0.5
def _check_score_count_pairing(independent_df, dependent_df, merged_df, *,
well_column='prc', record=None):
"""Fail loudly when the score and count tables describe different wells.
The two inputs are never paired file-to-file: each list is concatenated
and the two are joined on ``prc`` (``plateID_rowID_columnID``). So the
plate ID is the pairing key, and a plate ID that differs by one character
between the two sides silently produces an empty join.
That is not hypothetical. A legacy score CSV carries its plate in a
``plate`` column stamped ``pplate1``, while the sequencing counts carry
``plate1``. Before :func:`spacr.utils.correct_metadata` was fixed to
normalise that after the legacy promotion, the join returned zero rows and
the run continued for another two hundred lines before dying inside a plot
with ``KeyError: 0`` -- an error naming neither the plates, the files, nor
the join.
:param independent_df: Count-table rows before the score/count join.
:param dependent_df: Score-table rows before the score/count join.
:param merged_df: Rows retained by the score/count join.
:param well_column: Column containing the unique well identifier.
:param record: Optional mutable mapping that receives the matched and
unmatched well counts for the persisted run summary.
:raises ValueError: when the join is empty, or retains less than
:data:`_MINIMUM_PAIRED_WELL_FRACTION` of the smaller input's wells.
"""
def _plates(frame):
"""Return sorted plate prefixes parsed from the frame's well column."""
if well_column not in frame.columns:
return []
return sorted(frame[well_column].astype(str).str.split('_').str[0]
.dropna().unique())
count_wells = independent_df[well_column].nunique() if \
well_column in independent_df.columns else 0
score_wells = dependent_df[well_column].nunique() if \
well_column in dependent_df.columns else 0
score_plates = _plates(dependent_df)
count_plates = _plates(independent_df)
matched = merged_df[well_column].nunique() if \
well_column in merged_df.columns else 0
comparable = min(score_wells, count_wells)
if comparable and matched / comparable >= _MINIMUM_PAIRED_WELL_FRACTION:
unused_counts = count_wells - matched
unused_scores = score_wells - matched
if record is not None:
record["wells_paired"] = int(matched)
record["wells_unpaired_counts"] = int(unused_counts)
record["wells_unpaired_scores"] = int(unused_scores)
if unused_counts or unused_scores:
paired_label = "well" if matched == 1 else "wells"
count_label = "well" if unused_counts == 1 else "wells"
score_label = "well" if unused_scores == 1 else "wells"
print(
f"Paired {matched} {paired_label}. {unused_counts} "
f"count-table {count_label} and {unused_scores} score-table "
f"{score_label} had no matching identifier and were "
f"excluded from the "
f"regression.")
return
shared = sorted(set(score_plates) & set(count_plates))
detail = (
f"score wells: {score_wells} on plates {score_plates}\n"
f" count wells: {count_wells} on plates {count_plates}\n"
f" shared plates: {shared or 'NONE'}\n"
f" paired wells: {matched}"
)
if matched == 0:
raise ValueError(
f"The score and count tables have no well in common, so the "
f"regression has nothing to fit.\n\n"
f" {detail}\n\n"
f"They are joined on prc = plateID_rowID_columnID, so the plate "
f"ID is what pairs them -- the ORDER you listed the files in does "
f"not matter, and the two lists need not be the same length. Make "
f"the plate IDs agree: give every input a plateID column with "
f"matching values, or state the pairing explicitly.")
raise ValueError(
f"Only {matched} of {comparable} pairable wells "
f"({matched / comparable:.1%}) found a partner, which is below the "
f"{_MINIMUM_PAIRED_WELL_FRACTION:.0%} required. The two inputs are "
f"probably describing different plates or different well layouts.\n\n"
f" {detail}\n\n"
f"Continuing would fit the model on whichever wells happened to "
f"overlap and report it as the whole screen.")
def _identifiability_warning(data, settings, *, well_column='prc',
guide_column='grna', level='grna'):
"""Warn when a fit is about to be run on too few wells.
Returns the warning text, or ``None`` when the design is fine. Kept
separate from :func:`resolve_auto_inference` because this one never
changes what runs -- it only makes sure the user cannot miss what they
are about to get.
Count terms at the requested fit level because guide- and gene-level
designs can have different widths. Return a warning only when the number
of estimated intercept, block, and identifier terms is at least the
number of analyzed wells.
:param level: ``'grna'`` (default) or ``'gene'`` -- which fit is about to
run, and therefore which identifiers are the parameters.
"""
identifier = 'gene' if str(level).strip().lower() == 'gene' else guide_column
try:
n_wells = int(data[well_column].nunique())
n_terms = int(data[identifier].nunique())
except (KeyError, TypeError):
return None
blocks = 0
block_column = str(settings.get('guide_permutation_block', 'plateID'))
if block_column in getattr(data, 'columns', ()):
blocks = max(int(data[block_column].nunique()) - 1, 0)
parameters = 1 + blocks + n_terms
if n_wells > parameters:
return None
return (
"\n"
" ###############################################################\n"
" # WARNING: this fit is saturated or not identifiable. #\n"
" ###############################################################\n"
f" {n_wells} analysed wells are being used to estimate "
f"{parameters} parameters\n"
f" ({n_terms} {identifier}s + intercept + {blocks} block terms).\n"
"\n"
" With at least as many parameters as wells, the model has no\n"
" residual degrees of freedom and may also be rank deficient.\n"
" Individual guide coefficients, standard errors and P values\n"
" cannot be interpreted reliably.\n"
"\n"
" Set inference='nonparametric' to test each guide as a\n"
" marginal association, wells reshuffled within each plate,\n"
" coefficients simultaneously, or inference='auto' to let spaCR\n"
" choose. The design\n"
" diagnostics written beside the results show the rank, the\n"
" residual degrees of freedom and the collinear guide pairs.\n")
def _usable_nuisance_columns(data, settings) -> list:
"""The nuisance columns that are actually in the frame, said out loud.
`guide_nuisance_columns` defaults to row and column, which every spaCR
screen has and an imported table might not. `_nuisance_design` raises on
an absent column -- correct for one the user typed, wrong for one that
arrived as a default -- so the filtering happens here.
SAID, NOT SILENT. A user who believes position was removed and reads a
p-value computed without removing it has been told something false by
omission, and the exchangeability the permutation rests on is exactly
what those columns were there to protect.
"""
wanted = [str(c) for c in (settings.get('guide_nuisance_columns') or [])]
if not wanted:
return []
have = set(map(str, getattr(data, 'columns', ())))
usable = [c for c in wanted if c in have]
missing = [c for c in wanted if c not in have]
if usable:
from .guide_permutation import _nuisance_design
block = str(settings.get('guide_permutation_block', 'plateID'))
while usable:
try:
_nuisance_design(data, block, usable)
break
except ValueError as exc:
if "rank deficient" not in str(exc):
break
dropped = usable.pop()
print(f"■ guide_nuisance_columns: {dropped!r} is collinear "
f"with {block!r} on this layout -- every level of one "
f"determines a level of the other -- so it cannot be "
f"removed separately. Dropped; {block!r} already "
f"absorbs it.")
except Exception: # noqa: BLE001
break
if missing:
print(f"■ guide_nuisance_columns named {len(missing)} column(s) this "
f"table does not have: {', '.join(missing)}. They are not "
f"removed before the permutation, so any structure they carry "
f"stays in the residual the shuffle treats as noise.")
return usable
def _report_exchangeability(data, outcome_column, settings, destination):
"""Measure and report whether the within-block shuffle is defensible.
A COURTESY, NOT A PRECONDITION -- the same rule the montage pre-flight
follows. It must never be the reason a run that produced results fails
to report them, so every step is inside the guard.
:param data: the merged per-well table.
:param outcome_column: one phenotype column name.
:param settings: the run's settings.
:param destination: the run's results folder. When given, the report and
a residual-by-position figure per block are written to its
``regression_qc`` folder, as a parametric run's QC is; the report's
``'qc'`` key holds what was written.
:returns: the :func:`spacr.permutation_qc.block_residual_report`, or
``None`` when the check could not run.
"""
try:
from .guide_permutation import (_nuisance_design, _residualize,
prepare_long_guide_data)
from .permutation_qc import (block_residual_report,
exchangeability_verdict,
write_permutation_qc)
block = str(settings.get('guide_permutation_block', 'plateID'))
nuisance = _usable_nuisance_columns(data, settings)
wanted = list(dict.fromkeys([*nuisance, 'rowID', 'columnID']))
present = [c for c in wanted
if c in getattr(data, 'columns', ())]
_f, outcomes, _m = prepare_long_guide_data(
data, outcome_column, block_column=block,
nuisance_columns=present)
y = pd.to_numeric(outcomes[outcome_column],
errors='coerce').to_numpy(dtype=float)
basis, _r = np.linalg.qr(
_nuisance_design(outcomes, block, nuisance), mode='reduced')
residuals = _residualize(y, basis)
positions = {c: outcomes[c] for c in present
if c in outcomes.columns and c != block}
report = block_residual_report(
residuals, outcomes[block], positions)
verdict = exchangeability_verdict(report)
if destination:
try:
report['qc'] = write_permutation_qc(
destination, outcome_column, residuals,
outcomes[block], positions, report, verdict,
removed=[c for c in nuisance if c != block])
if report['qc'].get('figure'):
print(f"Permutation QC written to {report['qc']['dir']}")
except Exception as error: # noqa: BLE001
print(f"Permutation QC could not be written: "
f"{type(error).__name__}: {error}")
if verdict['ok']:
print(f"Exchangeability: nothing found. Durbin-Watson "
f"{report['durbin_watson']:.2f} over {report['n']:,} well(s) "
f"in {report['blocks']} block(s), and no position column "
f"explains the residual.")
return report
print("■ Exchangeability: the within-block shuffle is questionable.")
for finding in verdict['findings'][:4]:
print(f" {finding}")
if verdict['remedy']:
print(f" -> {verdict['remedy']}")
return report
except Exception: # noqa: BLE001
LOG.debug("could not report exchangeability", exc_info=True)
return None
[docs]
def resolve_regression_src(requested, automatic):
"""Resolve the root directory used for regression output.
A blank ``requested`` value selects ``automatic``. An existing requested
directory is used directly. If only the final path component is missing,
that directory is created; missing parent directories are never created.
A requested file, an unavailable parent, or a directory-creation error
returns the automatic location with an explanatory message.
:param requested: Requested output directory, or ``None``/blank to use
the automatic location.
:param automatic: Existing fallback directory, normally the directory
containing the first count table.
:returns: A ``(path, message)`` tuple. ``message`` is ``'automatic'``
when no override was requested; otherwise it describes the selected
directory or the reason for falling back.
"""
if not isinstance(requested, str) or not requested.strip():
return automatic, 'automatic'
wanted = os.path.abspath(os.path.expanduser(requested.strip()))
if os.path.isdir(wanted):
return wanted, f"Regression output directory: {wanted}."
if os.path.exists(wanted):
return automatic, (
f"The configured regression output path {wanted} is not a "
f"directory. Results will be written to the automatic location "
f"{automatic}.")
parent = os.path.dirname(wanted)
if os.path.isdir(parent):
try:
os.mkdir(wanted)
except OSError as error:
return automatic, (
f"The regression output directory {wanted} could not be "
f"created ({error.strerror or type(error).__name__}). "
f"Results will be written to the automatic location "
f"{automatic}.")
return wanted, f"Created regression output directory: {wanted}."
return automatic, (
f"The regression output directory {wanted} was not created because "
f"its parent directory {parent} does not exist. Results will be "
f"written to the automatic location {automatic}.")
def _run_guide_permutation_analysis(data, outcome, destination, settings):
"""Run and persist the marginal guide analysis.
This is the ``perform_regression`` branch used when
``analysis_mode='guide_permutation'``. Keeping it as a top-level function
makes the correction and output contract testable without replaying score
aggregation and sequencing QC.
:returns: The long results, selected support family, significant rows, and
a mapping of every artifact written by the analysis.
"""
from .guide_permutation import (
analyse_long_guide_table,
plot_guide_permutation_volcano,
save_guide_permutation_results,
)
thresholds = settings.get('guide_min_wells', [1, 2, 3, 4])
if isinstance(thresholds, (int, np.integer)):
thresholds = [int(thresholds)]
thresholds = sorted({int(value) for value in thresholds})
if not thresholds or any(value < 1 for value in thresholds):
raise ValueError('guide_min_wells must contain positive integers')
primary = settings.get('guide_primary_min_wells')
primary = thresholds[0] if _left_blank(primary) else int(primary)
if primary not in thresholds:
raise ValueError(
f'guide_primary_min_wells={primary} is not in '
f'guide_min_wells={thresholds}')
destination = os.path.abspath(os.path.expanduser(os.fspath(destination)))
os.makedirs(destination, exist_ok=True)
outcomes = [outcome] if isinstance(outcome, str) else list(outcome)
missing = [column for column in outcomes if column not in data.columns]
if missing:
raise ValueError(
f"dependent_variable names {missing} which are not columns of the "
f"merged table. Available: {sorted(data.columns)[:20]}")
per_object = str(settings.get('analysis_unit', 'well')).lower() != 'well'
unaggregated = settings.get('agg_type') is None
if per_object or unaggregated:
why = (f"analysis_unit={settings.get('analysis_unit')!r}"
if per_object else
f"agg_type is None (regression_type="
f"{settings.get('regression_type')!r} fits objects)")
raise ValueError(
f"analysis_mode='guide_permutation' tests each guide across "
f"WELLS, so it needs one row per well -- but "
f"{why} gives one row "
f"per object, and a well's phenotype then has many values. Set "
f"analysis_unit='well' (with an agg_type such as 'mean'), or "
f"choose analysis_mode='regression', which can model objects.")
results = analyse_long_guide_table(
data,
outcomes,
min_wells=thresholds,
block_column=str(settings.get('guide_permutation_block', 'plateID')),
nuisance_columns=_usable_nuisance_columns(data, settings),
n_permutations=int(settings.get('guide_permutations', 200000)),
random_state=int(settings.get('guide_permutation_seed', 0)),
multiple_testing=str(settings.get('multiple_testing_method', 'fdr_bh')),
alpha=float(settings.get('fdr_alpha', 0.05)),
presence_threshold=float(settings.get('guide_presence_threshold', 0.0)),
batch_size=int(settings.get('guide_permutation_batch_size', 500)),
statistic=str(settings.get('grna_statistic', 'pearson')),
)
for outcome_column in outcomes:
_report_exchangeability(data, outcome_column, settings, destination)
results = results.copy()
results['grna'] = results['guide']
results['feature'] = (
'fraction:grna[' + results['guide'].astype(str) + ']')
results['coefficient'] = results['standardized_marginal_effect']
results['p_value'] = results['permutation_p_value']
results['q_value'] = results['adjusted_p_value']
results['condition'] = label_control_condition(
results['feature'], results['grna'],
nc=settings.get('negative_control_id'),
pc=settings.get('positive_control_id'),
controls=settings.get('nontargeting_control_grnas'))
from .thresholds import coefficient_threshold
control_effects = results.loc[
(results['minimum_wells_threshold'] == primary)
& results['condition'].isin(('nc', 'control')), 'coefficient']
effect_threshold, effect_rule = coefficient_threshold(
control_effects,
method=settings.get('threshold_method', 'std'),
multiplier=settings.get('threshold_multiplier', 3.0),
centre=None)
print(f"Effect-size cut: {effect_rule}")
results['effect_size_threshold'] = (
np.nan if effect_threshold is None else float(effect_threshold))
results['passes_effect_size'] = (
True if effect_threshold is None
else results['coefficient'].abs() >= float(effect_threshold))
paths = dict(save_guide_permutation_results(
results, destination, prefix='guide_permutation'))
if settings.get('guide_permutation_plot', True):
single = len(outcomes) == 1
for response in outcomes:
for threshold in thresholds:
have = results.loc[
(results['outcome'] == response)
& (results['minimum_wells_threshold'] == int(threshold))]
if have.empty:
print(f"No guide reached {threshold} well(s) for "
f"{response!r}, so that panel of the "
f"guide_min_wells sweep is not drawn. The "
f"thresholds that did have guides are unaffected.")
continue
for suffix in ('pdf', 'png'):
stem = (f'guide_permutation_min_{threshold}_wells'
if single else
f'guide_permutation_{response}_min_'
f'{threshold}_wells')
key = (f'plot_min_{threshold}_{suffix}' if single else
f'plot_{response}_min_{threshold}_{suffix}')
paths[key] = plot_guide_permutation_volcano(
results,
outcome=response,
minimum_wells=threshold,
save_path=os.path.join(
destination, f'{stem}.{suffix}'),
effect_threshold=effect_threshold,
effect_threshold_label=effect_rule,
)
try:
from .guide_permutation import prepare_long_guide_data
from .regression_diagnostics import write_diagnostic_suite
fractions, well_outcomes, _metadata = prepare_long_guide_data(
data, outcomes,
block_column=str(settings.get('guide_permutation_block', 'plateID')),
nuisance_columns=list(settings.get('guide_nuisance_columns') or []))
for response in outcomes:
family = results.loc[
(results['outcome'] == response)
& (results['minimum_wells_threshold'] == primary)]
written = write_diagnostic_suite(
os.path.join(destination, 'diagnostics'),
fractions=fractions,
block=well_outcomes[
str(settings.get('guide_permutation_block', 'plateID'))],
p_values=family['permutation_p_value'].to_numpy(),
adjusted=family['adjusted_p_value'].to_numpy(),
alpha=float(settings.get('fdr_alpha', 0.05)),
label=response if len(outcomes) > 1 else '',
presence_threshold=float(
settings.get('guide_presence_threshold', 0.0)),
)
prefix = f'{response}_' if len(outcomes) > 1 else ''
paths.update({f'{prefix}{key}': value
for key, value in written.items()})
except Exception as error: # noqa: BLE001 - diagnostics are advisory
print(f"Regression diagnostics were skipped: "
f"{type(error).__name__}: {error}")
primary_table = results.loc[
results['minimum_wells_threshold'] == primary
].copy()
called = primary_table['significant'].astype(bool)
wide_enough = primary_table['passes_effect_size'].astype(bool)
significant = primary_table.loc[called & wide_enough].copy()
if effect_threshold is not None:
print(f"Effect-size cut removed {int((called & ~wide_enough).sum())} "
f"of {int(called.sum())} guides that passed correction but "
f"whose effect is narrower than {float(effect_threshold):.3g}.")
wanted_level = str(settings.get('level') or 'both').strip().lower()
wants_gene = wanted_level in ('gene', 'both')
if 'guide_permutation_gene_level' in settings:
wants_gene = bool(settings.get('guide_permutation_gene_level'))
gene_primary = None
if wants_gene:
try:
from .guide_permutation import analyse_long_gene_table
gene_results = analyse_long_gene_table(
data, outcomes,
min_wells=thresholds,
block_column=str(settings.get('guide_permutation_block',
'plateID')),
nuisance_columns=list(settings.get('guide_nuisance_columns')
or []),
n_permutations=int(settings.get('guide_permutations', 200000)),
random_state=int(settings.get('guide_permutation_seed', 0)),
multiple_testing=str(settings.get('multiple_testing_method',
'fdr_bh')),
alpha=float(settings.get('fdr_alpha', 0.05)),
presence_threshold=float(
settings.get('guide_presence_threshold', 0.0)),
batch_size=int(settings.get('guide_permutation_batch_size',
500)),
)
gene_results['feature'] = (
'gene_fraction:gene[' + gene_results['gene'].astype(str) + ']')
gene_results['grna'] = None
gene_results['coefficient'] = gene_results[
'standardized_marginal_effect']
gene_results['p_value'] = gene_results['permutation_p_value']
gene_results['q_value'] = gene_results['adjusted_p_value']
gene_results['condition'] = label_control_condition(
gene_results['feature'], gene_results['gene'],
nc=settings.get('negative_control_id'),
pc=settings.get('positive_control_id'),
controls=settings.get('nontargeting_control_grnas'))
gene_primary = gene_results.loc[
gene_results['minimum_wells_threshold'] == primary].copy()
print(f"Gene pass: {len(gene_primary)} genes tested as sets in "
f"the primary >={primary}-well family, corrected as their "
f"OWN BH family beside the {len(primary_table)} guides.")
except Exception as error: # noqa: BLE001 - the guide pass still stands
print(f"The gene-level permutation pass could not run: "
f"{type(error).__name__}: {error}. results_gene.csv will be "
f"empty; the guide results are unaffected.")
gene_primary = None
compatibility = {
'results': os.path.join(destination, 'results.csv'),
'results_grna': os.path.join(destination, 'results_grna.csv'),
'results_gene': os.path.join(destination, 'results_gene.csv'),
'significant': os.path.join(destination, 'results_significant.csv'),
}
levelled = primary_table.copy()
levelled['level'] = 'grna'
gene_rows = None
if gene_primary is not None and len(gene_primary):
gene_rows = gene_primary.copy()
if wanted_level == 'gene' and gene_rows is not None:
combined = gene_rows
elif gene_rows is not None and wanted_level != 'grna':
combined = pd.concat([levelled, gene_rows], ignore_index=True,
sort=False)
else:
combined = levelled
combined.to_csv(compatibility['results'], index=False)
primary_table.to_csv(compatibility['results_grna'], index=False)
(gene_primary if gene_primary is not None
else primary_table.iloc[0:0]).to_csv(
compatibility['results_gene'], index=False)
significant.to_csv(compatibility['significant'], index=False)
paths.update(compatibility)
return {
'analysis_mode': 'guide_permutation',
'results': combined,
'families': results,
'gene_results': gene_primary,
'primary': primary_table,
'significant': significant,
'primary_min_wells': primary,
'effect_size_threshold': effect_threshold,
'effect_size_rule': effect_rule,
'paths': {key: str(path) for key, path in paths.items()},
}
#: Settings a run chose for itself because the user left them unset. Filled
#: by :func:`perform_regression` as each is derived, and printed once both
#: are known -- the settings table is rendered before either exists.
_AUTOMATIC_SETTINGS: dict = {}
def _perform_regression_set_paths(settings):
"""Resolve and reserve a run's result directory and output paths.
:param settings: Normalized regression settings; updated with resolved
``src`` and ``_regression_folder``.
:returns: Results, gene, guide, and significant CSV paths, followed by the
results directory and first count-data path.
"""
csv_path = settings['count_data'][0]
from . import tabular
remote_count = tabular._backend_of(csv_path) == 'postgres'
automatic = '' if remote_count else os.path.dirname(csv_path)
requested = settings.get('src')
blank = str(requested or '').strip() in ('', 'path', '/path', '/path/to/src')
src, how = ('', '') if remote_count and (
blank or tabular._backend_of(requested) == 'postgres') else \
resolve_regression_src(requested, automatic)
if remote_count and not src:
from .qt.i18n import tr
raise ValueError(tr(
'A local regression output directory is required for PostgreSQL counts.'))
settings['src'] = src
if how != 'automatic':
print(how)
kind = results_folder_kind(settings)
res_folder = _next_results_folder(os.path.join(src, 'results'), kind)
_stage(settings, "placing the results folder")
settings["_regression_folder"] = res_folder
os.makedirs(res_folder, exist_ok=True)
results_filename = 'results.csv'
results_filename_gene = 'results_gene.csv'
results_filename_grna = 'results_grna.csv'
hits_filename = 'results_significant.csv'
results_path=os.path.join(res_folder, results_filename)
results_path_gene=os.path.join(res_folder, results_filename_gene)
results_path_grna=os.path.join(res_folder, results_filename_grna)
hits_path=os.path.join(res_folder, hits_filename)
return results_path, results_path_gene, results_path_grna, hits_path, res_folder, csv_path
[docs]
def results_folder_kind(settings) -> str:
"""What a run's results folder is NAMED after.
The inference method when it decides the answer, and the regression type
otherwise. Under `analysis_mode='guide_permutation'` the regression type
is never read -- ols and mixed produce byte-identical results -- so a
folder called `ridge` would name something the run did not do.
PUBLIC, AND THE ONLY COPY. A test that re-derived this rule went stale
when the rule changed and reported 39 missing CSVs while every run that
wrote them was fine, which is the failure the `results_dir` helper in
tests/test_cov_ml_perform_regression.py was already written to prevent
once. A suite pointing at the wrong file is worse than a silent one.
:param settings: run settings mapping, or ``None`` (treated as empty);
only ``analysis_mode`` and ``regression_type`` are read.
:returns: ``'guide_permutation'``, ``'auto'`` when no regression type is
set, or the regression type as a string.
"""
settings = settings or {}
if settings.get('analysis_mode') == 'guide_permutation':
return 'guide_permutation'
if settings.get('regression_type') is None:
return 'auto'
return str(settings['regression_type'])
def _next_results_folder(root, kind, limit=1000):
"""``<root>/<kind>``, or ``<kind>_1``, ``<kind>_2`` ... if taken.
A run never writes on top of an earlier one. The old fixed path meant
comparing two corrections, or re-running with one setting changed, left
only the last on disk with nothing said about it -- and the results the
user was looking at were not the results they thought.
A folder counts as taken when it EXISTS AND HAS ANYTHING IN IT. An empty
one is a directory somebody made and did not fill, and stepping past it
would strand it forever.
:param limit: stop after this many, rather than spinning if a filesystem
keeps answering "yes, that exists too".
"""
import os
base = os.path.join(root, str(kind))
for index in range(limit):
candidate = base if index == 0 else f"{base}_{index}"
try:
if not os.path.isdir(candidate) or not os.listdir(candidate):
return candidate
except OSError:
continue
return f"{base}_{limit}"
def _bracketed_identifier(pattern, text):
"""The id inside ``pattern``'s bracket, or ``None`` when there is none.
The inline version of this was ``re.search(...).group(1) if 'grna' in x
else None``, which assumes a term containing the word also contains the
bracket. It does not: a mixed fit's variance component is named
``'grna Var'``, so the search returns None and ``.group(1)`` raises
``AttributeError`` on a run that had already fitted its model.
"""
match = re.search(pattern, str(text))
return match.group(1) if match else None
def _annotate_level_coefficients(coef_df, n_grna, n_gene):
"""Attach the guide / gene id and the per-id row counts to ONE fit's table.
:param coef_df: one level's coefficient table, straight out of
:func:`regression`.
:param n_grna: value_counts frame, one row per guide.
:param n_gene: value_counts frame, one row per gene.
:returns: a new frame with ``grna``, ``gene``, ``n_grna`` and ``n_gene``.
"""
coef_df = coef_df.copy()
coef_df['grna'] = coef_df['feature'].map(
lambda value: _bracketed_identifier(r'grna\[(.*?)\]', value))
coef_df['gene'] = coef_df['feature'].map(
lambda value: _bracketed_identifier(r'gene\[(.*?)\]', value))
carried = dict(getattr(coef_df, "attrs", {}) or {})
coef_df = coef_df.merge(n_grna, how='left', on='grna',
validate='many_to_one')
coef_df = coef_df.merge(n_gene, how='left', on='gene',
validate='many_to_one')
if carried:
coef_df.attrs.update(carried)
return coef_df
def _level_control_rows(frame, level, controls):
"""The control rows of ONE fit's table, matched at that fit's own level.
``settings['nontargeting_control_grnas']`` names GUIDES. The guide fit matches them whole,
exactly as it always has. The gene fit has no guide column at all -- every
``gene_fraction:gene[...]`` row carries ``grna=None`` -- so matching the
same list there selects nothing and the gene table silently gets no
effect-size cut. A control guide identifies its gene by spaCR's own rule
(:func:`spacr.hits.gene_of`: truncate at the first underscore), so the
gene fit matches on that.
"""
if not (controls or []):
return frame.iloc[0:0]
from .control_names import matches, resolve_controls
guides = frame['grna'] if 'grna' in frame.columns else frame.index
library = [str(g) for g in pd.Series(guides).astype(str).unique()]
genes = frame['gene'] if 'gene' in frame.columns else None
specs = resolve_controls(controls, names=library)
if not specs:
return frame.iloc[0:0]
keep = None
for spec in specs:
if level == 'gene':
from .control_names import GENE, ControlSpec
spec = ControlSpec(spec.typed, GENE,
spec.value.split('_')[0] if not spec.is_gene
else spec.value, spec.prefix)
mask = matches(spec, frame['gene'].astype(str),
frame['gene'].astype(str))
else:
mask = matches(spec, pd.Series(guides).astype(str), genes)
keep = mask if keep is None else (keep | mask)
return frame.loc[keep.to_numpy()]
#: What `annotation_source` calls the bundled, offline Toxoplasma path.
BUNDLED_ANNOTATION = "toxoplasma"
def _annotation_source(settings) -> str:
"""Which organism's annotation this run asked for, or "" for none.
`annotation_source` is the one setting that says it. A dict that still
carries the retired `Toxoplasma` or `toxo` switch is read through
:func:`spacr.settings._fold_toxoplasma` on a copy, so a caller that
hands this module a raw, unfolded dict gets the same answer the
regression defaults would give it: a name wins, and otherwise true
means the bundled tables and false means no annotation.
"""
from .settings import _fold_toxoplasma
folded = _fold_toxoplasma(dict(settings or {}), quiet=True)
return str(folded.get('annotation_source', '') or '').strip()
def _toxoplasma_is_on(settings) -> bool:
"""Whether this run's annotation is the bundled *Toxoplasma* one.
It gates the Toxoplasma-only figures -- the hyperLOPIT volcano and the
GT1/ME49 phenotype and expression reports -- which mean nothing on
another organism's screen. Before 2026-09-19 it read the `Toxoplasma`
switch, which defaulted on, so a run annotated with 'human' still drew
them. The name in `annotation_source` decides now, through the same
resolver the annotation itself uses.
"""
source = _annotation_source(settings)
if not source:
return False
from .uniprot import resolve
return resolve(source).kind == "bundled"
def _annotation_cache(settings):
"""Where a UniProt answer is kept, so a rerun needs no network."""
src = settings.get('src')
if isinstance(src, (list, tuple)):
src = src[0] if src else None
if not src:
return None
return os.path.join(str(src), 'annotation_cache')
def _call_level_hits(coef_df, level, settings, regression_type,
merged_df, dependent_variable, bootstrap=None):
"""Correct one fit within itself and call that fit's hits.
Treat guide and gene fits as separate multiple-testing families. They use
the same wells and gene regressors are sums of guide regressors, so pooling
both levels would count correlated hypotheses as independent tests.
:param coef_df: one fit's annotated coefficient table.
:param level: ``'grna'`` or ``'gene'`` -- which fit this is.
:param regression_type: the backend that produced it.
:param merged_df: the pre-clean long frame, for the lasso bootstrap.
:param bootstrap: ``perform_regression``'s ``bootstrap_selection_frequencies``
closure. It is defined inside that function, so it cannot be looked up
from here and the penalised backends need it passed in.
:returns: ``(coef_df, significant, reg_threshold, effect_rule)``.
"""
from .thresholds import coefficient_threshold
coef_df = coef_df.copy()
reg_threshold = 0
effect_rule = 'no effect-size cut'
if settings['nontargeting_control_grnas'] is not None:
control_coef_df = _level_control_rows(
coef_df, level, settings['nontargeting_control_grnas'])
measured_threshold, threshold_rule = coefficient_threshold(
control_coef_df['coefficient'],
method=settings['threshold_method'],
multiplier=settings['threshold_multiplier'],
centre=None)
effect_rule = threshold_rule
print(f"Effect-size cut ({level}): {threshold_rule}")
reg_threshold = (0 if measured_threshold is None
else float(measured_threshold))
else:
print(f"Effect-size cut ({level}): no control gRNAs were named, so "
f"there is none; a hit is the corrected P value alone.")
if regression_type in NO_P_VALUE_TYPES and bootstrap is None:
raise ValueError(
f"regression_type={regression_type!r} ranks features by bootstrap "
f"selection frequency and has no p-value to correct, so "
f"_call_level_hits needs perform_regression's "
f"bootstrap_selection_frequencies passed as `bootstrap`.")
if regression_type in NO_P_VALUE_TYPES:
n_boot = settings.get('lasso_n_boot', 200)
sel_threshold = settings.get('lasso_selection_threshold', 0.6)
formula = prepare_formula(
dependent_variable, random_row_column_effects=False,
block_screen=screen_is_blockable(merged_df), level=level,
model_plate_position=settings.get('model_plate_position', True))
cleaned_df = check_and_clean_data(merged_df.copy(), dependent_variable)
sel_df = bootstrap(
X=cleaned_df,
y=cleaned_df[dependent_variable],
formula=formula,
alpha=settings.get('alpha', 'auto'),
n_boot=n_boot,
random_state=0,
regression_type=regression_type,
l1_ratio=settings['l1_ratio'],
group_lasso_lambda=settings.get('group_lasso_lambda', 'auto'),
)
coef_df = coef_df.merge(sel_df, on='feature', how='left',
validate='one_to_one')
significant = coef_df[
(coef_df['coefficient'] != 0)
& (coef_df['selection_frequency'] >= sel_threshold)
].copy()
significant = significant.sort_values(
by='coefficient', key=lambda c: c.abs(), ascending=False,
)
significant = significant[~significant['feature'].str.contains('row|column')]
return coef_df, significant, reg_threshold, effect_rule
from .multiple_testing import adjust_p_values, canonical_method
method = canonical_method(settings.get('multiple_testing_method',
'fdr_bh'))
alpha = float(settings.get('fdr_alpha', 0.05))
cut_alpha = float(settings.get('p_threshold_alpha', alpha) or alpha)
cut_kind = str(settings.get('p_threshold_kind', 'adjusted')).strip().lower()
cut_column = 'p_value' if cut_kind == 'raw' else 'q_value'
from .hits import tested_family
tested = pd.Series(tested_family(coef_df['feature']),
index=coef_df.index)
tested &= coef_df['p_value'].notna()
if 'term_type' in coef_df.columns:
tested &= coef_df['term_type'].eq(TERM_FIXED)
coef_df['q_value'] = np.nan
coef_df['multiple_testing_method'] = method
if tested.any():
adjusted, _rejected = adjust_p_values(
coef_df.loc[tested, 'p_value'].to_numpy(dtype=float),
method=method, alpha=alpha)
coef_df.loc[tested, 'q_value'] = adjusted
raw_hits = int((coef_df.loc[tested, 'p_value'] <= alpha).sum())
corrected_hits = int((coef_df.loc[tested, 'q_value'] < alpha).sum())
print(f"Multiple testing ({level}): {method} across {int(tested.sum())} "
f"tested coefficients at alpha={alpha:g} — {raw_hits} pass the raw "
f"P value, {corrected_hits} pass correction.")
if cut_kind == 'raw' or cut_alpha != alpha:
print(f" Calling hits on the {cut_kind} P at {cut_alpha:g}"
+ (", NOT corrected for multiple testing."
if cut_kind == 'raw' else "."))
significant = coef_df.loc[coef_df[cut_column] < cut_alpha].copy()
coef_df['effect_size_threshold'] = (
np.nan if not reg_threshold else abs(float(reg_threshold)))
coef_df['effect_size_rule'] = effect_rule
significant = significant.assign(
effect_size_threshold=(np.nan if not reg_threshold
else abs(float(reg_threshold))),
effect_size_rule=effect_rule)
if reg_threshold:
wide_enough = (significant['coefficient'].abs()
>= abs(reg_threshold))
called = len(significant)
significant = significant.loc[wide_enough].copy()
print(f"Effect-size cut ({level}) removed {called - len(significant)} "
f"of {called} coefficients that passed correction but whose "
f"effect is narrower than {abs(reg_threshold):.3g}.")
significant = significant.sort_values(
by='coefficient', ascending=False)
significant = significant[~significant['feature'].str.contains('row|column')]
return coef_df, significant, reg_threshold, effect_rule
def _stage(settings, name):
"""Record the current fit stage, announce it, and never raise.
Fall back to storing ``_regression_stage`` in the settings mapping when
resource measurement is unavailable.
THE ANNOUNCEMENT IS THE POINT AS MUCH AS THE RECORD. Reading the counts
and fitting the model each take minutes on a four-plate screen, and a step
that prints nothing while it runs cannot be told apart from a step that
has hung -- which is how a working run comes to be reported as a dead one.
The recorded resident size goes on the same line, because the other thing
a long silent step invites is a guess about memory.
"""
reading = {}
try:
from .fit_resources import record_stage
reading = record_stage(settings, name)
except Exception: # noqa: BLE001
try:
settings["_regression_stage"] = str(name)
except Exception: # noqa: BLE001
pass
try:
rss = reading.get("rss") if isinstance(reading, dict) else None
note = f" (resident {rss / 1e9:.1f} GB)" if rss else ""
print(f"Regression: {name}{note}.", flush=True)
except Exception: # noqa: BLE001
pass
return reading
#: Panels that need a fitted object exposing residuals, and the models that
#: cannot supply one. RRA is a rank statistic -- it never fits a linear
#: predictor, so "residual" has no meaning for it rather than being
#: unavailable. Naming them here rather than catching AttributeError keeps the
#: REASON: a missing QQ plot and an inapplicable one look identical to a
#: reader, and only the inapplicable case is valid.
RESIDUAL_FREE_MODELS: dict = {
"rra": ("Robust Rank Aggregation is a rank statistic: it ranks guides "
"within each well and aggregates those ranks, so it never forms a "
"linear predictor and there is no residual to plot."),
"horseshoe": ("The horseshoe fit is sampled rather than solved, so it has "
"a posterior rather than one set of fitted values."),
}
def _diagnostic_inputs(model):
"""``(observed, fitted, design)`` from a fitted model, or ``(None,)*3``.
Duck-typed on purpose. statsmodels results expose ``fittedvalues``,
``resid`` and ``model.exog``; the backends spaCR wraps do not share a base
class, so asking what an object HAS is the only question that works across
all of them.
"""
fitted = getattr(model, "fittedvalues", None)
resid = getattr(model, "resid", None)
if fitted is None or resid is None:
return None, None, None
try:
observed = np.asarray(fitted, dtype=float) + np.asarray(resid,
dtype=float)
except Exception: # noqa: BLE001
return None, None, None
design = getattr(getattr(model, "model", None), "exog", None)
return observed, np.asarray(fitted, dtype=float), design
def _diagnostic_screen_design(data, settings):
"""Return the well-by-guide matrix and aligned block labels for QC.
``perform_regression`` works with a long table because the historical OLS
formula has one fitted row per well-guide pair. Identifiability, however,
is a question about independent wells and guide predictors. Passing that
long mixed-type table straight to ``design_report`` both miscounted the
observations and failed while converting ``prc`` to float. This adapter
performs the same sum-and-zero-fill pivot used by the permutation path and
verifies that a well has exactly one block label.
A caller that already supplies a numeric wide matrix keeps the original
API: it is returned unchanged with no inferred block.
"""
if not isinstance(data, pd.DataFrame):
return data, None
required = {schema.PRC_KEY, "grna", "fraction"}
if not required.issubset(data.columns):
return data, None
frame = data.loc[:, [schema.PRC_KEY, "grna", "fraction"]].copy()
if frame[[schema.PRC_KEY, "grna"]].isna().any().any():
raise ValueError("well and guide identifiers must not contain missing values")
frame["fraction"] = pd.to_numeric(frame["fraction"], errors="raise")
values = frame["fraction"].to_numpy(dtype=float)
if not np.isfinite(values).all():
raise ValueError("guide fractions must be finite")
if np.any(values < 0):
raise ValueError("guide fractions must be non-negative")
wide = frame.pivot_table(
index=schema.PRC_KEY, columns="grna", values="fraction",
aggfunc="sum", fill_value=0.0,
).sort_index()
wide.columns = wide.columns.astype(str)
block = None
block_column = str(settings.get("guide_permutation_block") or
schema.PLATE_KEY)
if block_column in data.columns:
labels = data.loc[:, [schema.PRC_KEY, block_column]].copy()
counts = labels.groupby(schema.PRC_KEY, sort=False)[block_column].nunique(
dropna=False)
inconsistent = counts > 1
if inconsistent.any():
example = inconsistent.index[inconsistent][0]
raise ValueError(
f"block labels are not constant within well {example!r}")
block = (labels.drop_duplicates(schema.PRC_KEY)
.set_index(schema.PRC_KEY)[block_column]
.reindex(wide.index))
if block.isna().any():
raise ValueError("block labels are missing for one or more wells")
return wide, block
def _write_regression_diagnostics(res_folder, fractions, fits, settings):
"""Write the diagnostic suite for a completed fit.
THE DESIGN REPORT IS UNCONDITIONAL. It needs no fit at all -- only the
well-by-guide matrix -- so it is available for every model including RRA,
and it is the one that would have caught the failure
:mod:`spacr.regression_diagnostics` was written for: 824 guides in 587
wells returning a confident P value for every guide out of a rank-deficient
matrix.
RESIDUAL PANELS SAY WHY WHEN THEY CANNOT BE DRAWN. `write_diagnostic_suite`
skips a block whose inputs are absent, silently, which is right for a
library and wrong here: the user asked for these plots "whenever possible",
and the interesting case is precisely when it is not possible. So a model
that cannot support residuals writes a note naming the reason beside the
panels that did run.
Never raises. A diagnostic that took the analysis down with it would be
worse than no diagnostic -- the numbers the user came for are already
computed by the time this runs.
"""
from . import regression_diagnostics as rd
if not res_folder:
return {}
destination = os.path.join(res_folder, "diagnostics")
written: dict = {}
try:
model, _coef, model_type = next(iter(fits.values()))
except Exception: # noqa: BLE001
model, model_type = None, str(settings.get("regression_type") or "")
observed, fitted, design = _diagnostic_inputs(model)
reason = RESIDUAL_FREE_MODELS.get(str(model_type).lower())
if observed is None and reason is None:
reason = (f"The {model_type or 'selected'} backend did not expose "
"fitted values and residuals, so the residual panels could "
"not be computed for this run.")
try:
fractions, block = _diagnostic_screen_design(fractions, settings)
written = dict(rd.write_diagnostic_suite(
destination, fractions=fractions, block=block,
observed=observed, fitted=fitted, design=design,
label=str(model_type or ""),
presence_threshold=float(
settings.get("guide_presence_threshold", 0.0) or 0.0)))
except Exception as error: # noqa: BLE001
print(f"Diagnostics could not be written: "
f"{type(error).__name__}: {error}")
return {}
if observed is None and reason:
note_path = os.path.join(destination, "residual_panels_not_available.txt")
try:
with open(note_path, "w", encoding="utf-8") as handle:
handle.write(reason + "\n")
written["residuals_unavailable"] = note_path
except OSError:
pass
print(f"Residual diagnostics were not computed: {reason}")
return written
def _write_regression_panel_packages(outcome, settings):
"""Build requested publication panels from this run's final CSV files."""
manifest = settings.get("regression_panel_manifest")
if manifest is None:
return None
settings["_regression_stage"] = "writing publication panel packages"
if not isinstance(outcome, dict):
raise TypeError("A regression panel manifest needs a mapping outcome")
raw_folder = (
outcome.get("res_folder")
or settings.get("_regression_folder")
or ""
)
if not raw_folder:
raise ValueError("A regression panel manifest needs a results folder")
res_folder = os.path.abspath(os.fspath(raw_folder))
settings["_regression_folder"] = res_folder
paths = outcome.get("paths")
paths = paths if isinstance(paths, dict) else {}
final_paths = {
"grna": os.path.abspath(os.fspath(
paths.get("results_grna")
or os.path.join(res_folder, "results_grna.csv")
)),
"gene": os.path.abspath(os.fspath(
paths.get("results_gene")
or os.path.join(res_folder, "results_gene.csv")
)),
}
missing = [path for path in final_paths.values() if not os.path.isfile(path)]
if missing:
raise FileNotFoundError(
"Regression panel source files do not exist: " + ", ".join(missing)
)
configured = settings.get("dependent_variable")
if isinstance(configured, str):
phenotypes = [configured]
elif isinstance(configured, (list, tuple)):
phenotypes = list(configured)
else:
phenotypes = []
if not phenotypes or any(not isinstance(value, str) or not value.strip()
for value in phenotypes):
raise ValueError(
"regression_panel_manifest needs one or more named "
"dependent_variable values"
)
phenotypes = [value.strip() for value in phenotypes]
fdr_alpha = float(settings.get("fdr_alpha", 0.05))
artifacts = {}
for level, path in final_paths.items():
table = tabular.read_table(path, report=None)
for phenotype in phenotypes:
selected = table
if len(phenotypes) > 1:
if "outcome" not in table.columns:
raise ValueError(
f"Multi-phenotype result {path} has no 'outcome' column"
)
selected = table.loc[table["outcome"].eq(phenotype)].copy()
if selected.empty:
raise ValueError(
f"Result {path} has no rows for phenotype {phenotype!r}"
)
artifacts[f"{phenotype}_{level}"] = {
"phenotype": phenotype,
"level": level,
"data": selected,
"run_artifact_path": path,
"res_folder": res_folder,
"dependent_variable": phenotype,
"fdr_alpha": fdr_alpha,
}
from .regression_panels import build_manifest_packages
return build_manifest_packages(
manifest,
artifacts,
os.path.join(res_folder, "publication_panels"),
)
#: What a completed run's resource record is called, beside its results.
FIT_RESOURCES_FILENAME = "fit_resources.txt"
def _write_fit_resources(outcome, settings):
"""Write per-stage and peak resource use for a successful fit.
Store the report beside the run results when a destination and
measurements are available. Return an empty string on missing data or any
measurement/write failure so resource reporting cannot fail the fit.
"""
try:
from .fit_resources import describe_resources, peak
folder = ""
if isinstance(outcome, dict):
folder = str(outcome.get("res_folder") or "")
if not folder:
folder = str(settings.get("_regression_folder", "") or "")
table = describe_resources(settings)
if not folder or not table or not os.path.isdir(folder):
return ""
high = peak(settings)
lines = ["WHAT THIS FIT COST", "==================", "",
"Recorded per stage as the run went. 'not measured' is not "
"zero:", "psutil absent, or no CUDA tensor allocated yet.",
"", table, ""]
if not high:
lines.append("No reading could be taken on this machine.")
path = os.path.join(folder, FIT_RESOURCES_FILENAME)
with open(path, "w", encoding="utf-8") as handle:
handle.write("\n".join(lines) + "\n")
return path
except Exception: # noqa: BLE001
return ""
def _warn_if_penalised_no_hits(settings, coef_df):
"""Explain why a penalised fit with no small P values is inconclusive."""
penalised = str(settings.get('regression_type', '')).lower() in (
'ridge', 'lasso', 'elasticnet')
if penalised and len(coef_df):
p_values = pd.to_numeric(coef_df.get('p_value'), errors='coerce')
if not (p_values < 0.05).any():
print(
f"\nNOTE: {settings['regression_type']} returned no "
f"coefficient below p=0.05. Its p-values are conservative "
f"by construction -- the standard error is unpenalised "
f"while the coefficient it is divided into has been shrunk "
f"-- so this is NOT evidence of no effect. Refit with "
f"regression_type='ols' (or 'rlm' for a robust check) "
f"before concluding anything from it.")
return True
return False
def _perform_regression(settings):
"""Regress per-well phenotype scores against gRNA / gene counts to identify hits from a pooled CRISPR screen.
Reads one or more score CSVs (from :func:`generate_ml_scores` or a
deep-learning classifier) and one or more sgRNA count CSVs (from
:func:`spacr.sequencing.generate_barecode_mapping`), aligns them on
plate / well, fits the requested regression model, merges metadata,
and emits volcano plots, plate heatmaps, gene phenotype plots and
GO enrichment reports.
:param settings: Settings dict, canonicalized via
:func:`spacr.settings.get_perform_regression_default_settings`.
Key entries:
- ``paired_data`` — ordered rows that explicitly pair one score CSV
with one sgRNA-count CSV. Plate identity comes from the files when
they agree, from the partner when only one declares it, or from the
pair-row order when neither does. Legacy ``score_data`` and
``count_data`` lists are migrated positionally with a visible log.
- ``dependent_variable`` — column of ``score_data`` to regress
(e.g. ``'pred'``, ``'recruitment'``,
``'pathogen_nucleus_shortest_distance'``).
- ``regression_type`` — any name in :data:`REGRESSION_TYPES`, or
``None`` to choose one from the response distribution. See
:func:`regression_model` for what each backend is for.
- ``analysis_mode='guide_permutation'`` — instead test plate-adjusted
marginal guide associations with empirical P values and apply
``multiple_testing_method`` (Benjamini--Hochberg by default) within
each requested ``guide_min_wells`` family.
- the per-model settings each backend reads — ``alpha``,
``l1_ratio``, ``cov_type``, ``quantile``, ``hinge_threshold``,
``hinge_n_boot``, ``huber_t``, ``random_row_column_effects``.
A setting the chosen type cannot read is refused rather than
ignored; see :data:`REGRESSION_SETTINGS_USED`.
- ``batch_correction`` — optional ``combat``, ``center``, ``zscore``,
``robust_zscore`` or reference-control ``control_center``
normalization of the dependent variable before well aggregation.
- ``fraction_threshold``, ``min_observations_per_hit``,
``metadata_files``,
``volcano``, ``heatmap_feature``.
:returns: Path to the merged, metadata-annotated results DataFrame
(also written to ``results/<score_source>/<regression_type>/
results.csv``). Related gene/gRNA CSVs and significance calls
are saved alongside.
:raises ValueError: if paired files declare incompatible plate IDs,
``dependent_variable`` is not a score column, ``regression_type`` is
unsupported, or a guide-permutation support family or correction
setting is invalid.
Example:
.. code-block:: python
from spacr.ml import perform_regression
settings = {
'paired_data': [{
'score': '/data/plate01/results/xgb_scores.csv',
'count': '/data/plate01/sequencing/counts.csv',
}],
'dependent_variable': 'pred',
'regression_type': 'mixed',
}
perform_regression(settings)
See Also:
:func:`generate_ml_scores` — produce the ``score_data`` input.
:func:`spacr.sequencing.generate_barecode_mapping` — produce the
``count_data`` input.
"""
from .plot import plot_plates, plot_data_from_csv
from .utils import merge_regression_res_with_metadata, save_settings, correct_metadata
from .settings import get_perform_regression_default_settings
from .toxo import custom_volcano_plot, plot_gene_phenotypes, plot_gene_heatmaps
def _perform_regression_read_data(settings):
"""Load paired inputs, validate the analysis, and return both frames."""
_stage(settings, "reading the input tables")
pairs, _migrated = normalize_regression_input_pairs(settings)
count_data_df, score_data_df, audit = \
load_regression_input_pairs(pairs)
settings['paired_data'] = pairs
settings['input_pair_audit'] = audit
print(f"Score data: {len(score_data_df)} rows from "
f"{len(settings['score_data'])} file(s)")
print(f"Count data: {len(count_data_df)} rows from "
f"{len(settings['count_data'])} file(s)")
print(f"Dependent variable: {len(score_data_df)}")
print(f"Independent variable: {len(count_data_df)}")
if settings['dependent_variable'] not in score_data_df.columns:
if not settings['dependent_variable'] == 'pathogen_nucleus_shortest_distance':
looks_like_counts = sorted(
{'grna', 'grna_name', 'count'}.intersection(
score_data_df.columns))
if looks_like_counts:
hint = (
f"\n\nThe score table has {looks_like_counts} and "
f"no score column, which is the shape of a COUNT "
f"file. The score and count inputs look swapped.")
else:
numeric = [
column for column in score_data_df.columns
if pd.api.types.is_numeric_dtype(
score_data_df[column])
and column not in {'plateID', 'rowID', 'columnID',
'fieldID', 'objectID', 'count'}]
hint = (f"\n\nColumns that could be the response: "
f"{numeric[:12]}") if numeric else ""
raise ValueError(
f"dependent_variable="
f"{settings['dependent_variable']!r} is not a column "
f"of the score table, which has "
f"{list(score_data_df.columns)[:15]}.{hint}")
_reject_impossible_probabilities(settings)
mode = str(settings.get('analysis_mode', 'regression')).strip().lower()
if mode not in {'regression', 'guide_permutation'}:
raise ValueError(
f"Unsupported analysis_mode {mode!r}; choose 'regression' "
"or 'guide_permutation'.")
settings['analysis_mode'] = mode
if mode == 'regression':
reg_type = settings['regression_type']
if reg_type is not None and reg_type not in REGRESSION_TYPES:
if reg_type in UNSUPPORTED_REGRESSION_TYPES:
raise ValueError(
f"Unsupported regression type {reg_type}: "
f"{UNSUPPORTED_REGRESSION_TYPES[reg_type]}")
print(f'Possible regression types: '
f'{list(REGRESSION_TYPES) + [None]}')
raise ValueError(f"Unsupported regression type {reg_type}")
_reconcile_random_row_column_effects(settings)
_reject_unused_run_settings(settings)
return count_data_df, score_data_df
def _count_variable_instances(df, column_1, column_2):
"""Return ``df`` and value-count tables for both named columns."""
for col in (column_1, column_2):
if col not in df.columns:
raise KeyError(
f"Column '{col}' not found in independent_df. "
f"Available columns: {list(df.columns)}"
)
n_grna = df[column_1].value_counts().reset_index()
n_grna.columns = [column_1, f"n_{column_1}"]
n_gene = df[column_2].value_counts().reset_index()
n_gene.columns = [column_2, f"n_{column_2}"]
return df, n_grna, n_gene
def _qc_plot(plot_settings):
"""Render one QC plot, reporting - not raising - on failure.
The QC tables written between these calls are data outputs, so a
plotting failure must not cost them. spacrGraph runs a group
comparison, which scipy rejects with "Must enter at least two input
sample vectors" on the very common single-plate run.
"""
try:
return plot_data_from_csv(settings=plot_settings)
except Exception as e:
print(f"Skipping QC plot {plot_settings['graph_name']!r}: {e}")
return None, None
def grna_metricks(df):
"""Return per-gRNA and per-well coverage counts derived from a long ``prc`` DataFrame.
:param df: DataFrame with ``prc``, ``grna`` and ``gene`` columns.
:returns: ``(final_grna_df, prc_gene_count_df)`` — per-gRNA
well counts and per-well distinct-gene counts.
"""
_assign_prc_parts(df)
grna_well_counts = (df.groupby(['grna', 'plateID'])['prc'].nunique().reset_index(name='grna_well_count'))
gene_well_counts = (df.groupby(['gene', 'plateID'])['prc'].nunique().reset_index(name='gene_well_count'))
unique_triplets = df[['grna', 'gene', 'plateID']].drop_duplicates()
merged_df = pd.merge(unique_triplets, grna_well_counts,
on=['grna', 'plateID'], how='left',
validate='many_to_one')
merged_df = pd.merge(merged_df, gene_well_counts,
on=['gene', 'plateID'], how='left',
validate='many_to_one')
final_grna_df = merged_df[['grna', 'plateID', 'grna_well_count', 'gene_well_count']]
prc_gene_count_df = (df.groupby('prc')['gene'].nunique().reset_index(name='gene_count'))
_assign_prc_parts(prc_gene_count_df)
return final_grna_df, prc_gene_count_df
def get_outlier_reference_values(df, outlier_col, return_col):
"""Return unique ``return_col`` values whose ``outlier_col`` falls outside 1.5*IQR.
:param df: Input DataFrame.
:param outlier_col: Numeric column screened for outliers.
:param return_col: Column whose distinct values are returned.
:returns: List of unique reference values for outlier rows.
"""
Q1 = df[outlier_col].quantile(0.05)
Q3 = df[outlier_col].quantile(0.95)
IQR = Q3 - Q1
lower_bound = Q1 - 1.5 * IQR
upper_bound = Q3 + 1.5 * IQR
outlier_mask = (df[outlier_col] < lower_bound) | (df[outlier_col] > upper_bound)
outliers = df.loc[outlier_mask, return_col]
outliers_ls = outliers.unique().tolist()
return outliers_ls
def bootstrap_selection_frequencies(X, y, formula, alpha='auto', n_boot=200,
random_state=None,
regression_type='lasso', l1_ratio=0.5,
group_lasso_lambda='auto'):
"""Return per-feature selection frequencies from a nonparametric bootstrap.
Output ranks features by how often their coefficient is non-zero
across resamples; this is a stability score, not a hypothesis test.
:param X: Long-form DataFrame; design matrix is built per resample
from ``formula`` for stable factor levels.
:param y: Response array aligned with ``X`` by index.
:param formula: Patsy formula for ``dmatrices``.
:param alpha: Regularisation strength; ``'auto'``/``None`` runs the
cross-validated estimator per resample.
:param n_boot: Number of bootstrap resamples. Default ``200``.
:param random_state: Seed for the resampling RNG.
:param regression_type: ``'lasso'``, ``'elasticnet'`` or
``'group_lasso'`` - the same penalty the reported coefficients
were fitted with, or the frequencies would describe a different
model from the one in ``results.csv``.
:param l1_ratio: ``elasticnet`` mix; ignored for ``'lasso'``.
:param group_lasso_lambda: the block penalty, for ``'group_lasso'``.
It is a separate argument from ``alpha`` because it is a separate
setting: ``alpha`` is not read by that backend at all, so a
resample fitted at ``alpha`` would be a different model from the
one whose coefficients this is ranking.
:returns: DataFrame with columns ``feature``,
``selection_frequency`` and ``mean_coefficient``.
:raises RuntimeError: if every resample fails to fit.
"""
rng = np.random.default_rng(random_state)
n = len(X)
use_cv = alpha is None or (isinstance(alpha, str) and alpha == 'auto')
def _estimator():
"""Return the configured sklearn lasso or elastic-net estimator."""
if regression_type == 'elasticnet':
return (ElasticNetCV(l1_ratio=l1_ratio, cv=5, max_iter=10000)
if use_cv else
ElasticNet(alpha=alpha, l1_ratio=l1_ratio, max_iter=10000))
return (LassoCV(cv=5, max_iter=10000) if use_cv
else Lasso(alpha=alpha, max_iter=10000))
y0, X0 = dmatrices(formula, data=X, return_type='dataframe')
feature_index = pd.Index(X0.columns)
blocks = (_design_column_groups(feature_index)
if regression_type == 'group_lasso' else None)
block_penalty = None
if blocks is not None:
from . import group_lasso as group_lasso_module
if _left_blank(group_lasso_lambda) or (
isinstance(group_lasso_lambda, str)
and group_lasso_lambda.strip().lower() == 'auto'):
block_penalty = group_lasso_module.choose_lambda(
np.asarray(X0, dtype=float),
np.asarray(y0, dtype=float).ravel(), blocks)
else:
block_penalty = float(group_lasso_lambda)
def _resample_coefficients(design, response):
"""Return group-lasso or sklearn coefficients for one resample."""
if blocks is not None:
from . import group_lasso as group_lasso_module
beta, _intercept, _converged = group_lasso_module.fit(
np.asarray(design, dtype=float),
np.asarray(response, dtype=float).ravel(),
blocks, lam=block_penalty)
return np.asarray(beta, dtype=float).ravel()
return np.asarray(_estimator().fit(design, response).coef_).ravel()
nonzero_counts = pd.Series(0.0, index=feature_index)
coef_sums = pd.Series(0.0, index=feature_index)
successful = 0
dropped = 0
last_failure = None
for _ in range(n_boot):
idx = rng.integers(0, n, size=n)
boot = X.iloc[idx].reset_index(drop=True)
try:
yb, Xb = dmatrices(formula, data=boot, return_type='dataframe')
except Exception as exc:
dropped += 1
last_failure = exc
continue
Xb = Xb.reindex(columns=feature_index, fill_value=0.0)
yb = np.asarray(yb).ravel()
coefs = pd.Series(_resample_coefficients(Xb, yb),
index=feature_index)
nonzero_counts += (coefs != 0).astype(float)
coef_sums += coefs
successful += 1
if successful == 0:
raise RuntimeError("All bootstrap resamples failed to fit. "
"Check the formula and ensure factor levels are not too sparse.")
if dropped:
LOG.warning(
"stability selection: %d of %d resamples produced no design "
"matrix (last error: %s). selection_frequency and "
"mean_coefficient below are over the remaining %d, not over "
"%d.", dropped, n_boot, last_failure, successful, n_boot)
return pd.DataFrame({
'feature': feature_index,
'selection_frequency': (nonzero_counts / successful).values,
'mean_coefficient': (coef_sums / successful).values,
})
settings = get_perform_regression_default_settings(settings)
count_data_df, score_data_df = _perform_regression_read_data(settings)
from .regression_layout import normalise_count_table_layout
count_data_df, resolved_input_layout = normalise_count_table_layout(
count_data_df,
layout=settings.get('independent_variable_layout', 'auto'),
guide_column=str(settings.get('count_grna_column') or 'grna'),
count_column=str(settings.get('count_value_column') or 'count'),
wide_predictor_columns=(settings.get('wide_predictor_columns') or None),
)
settings['independent_variable_layout_resolved'] = resolved_input_layout
print(
f"Independent-variable input: {resolved_input_layout}; normalized to "
f"long rows ({len(count_data_df):,} well-guide values)."
)
if "rowID" in count_data_df.columns:
count_data_df['rowID'] = (
count_data_df['rowID'].astype(str)
.str.rsplit(schema.KEY_SEPARATOR, n=1).str[-1]
)
if {'plateID', 'rowID', 'columnID'}.issubset(score_data_df.columns):
score_data_df['prc'] = (
_compose_prc_column(score_data_df)
)
if settings.get('verbose'):
print("score_data_df plateID counts:")
print(score_data_df['plateID'].value_counts())
print("count_data_df plateID counts:")
print(count_data_df['plateID'].value_counts())
results_path, results_path_gene, results_path_grna, hits_path, res_folder, csv_path = _perform_regression_set_paths(settings)
batch_method = str(
settings.get('batch_correction', 'none') or 'none'
).strip().lower()
if batch_method not in {'none', 'off', 'false'}:
dependent_variable = settings['dependent_variable']
if dependent_variable not in score_data_df.columns:
raise ValueError(
f"Batch correction cannot run because dependent_variable="
f"{dependent_variable!r} is not present in the score table. "
"Choose an existing score column or set batch_correction=none."
)
from .batch_correction import (
correct_from_metadata,
correction_kwargs,
write_report,
)
corrected, correction_report = correct_from_metadata(
score_data_df[[dependent_variable]],
score_data_df,
batch_covariate_column=settings.get('batch_covariate_column'),
batch_combat_mean_only=bool(
settings.get('batch_combat_mean_only', False)),
**correction_kwargs(settings),
)
report_path = write_report(
correction_report,
os.path.join(res_folder, 'batch_correction.json'),
)
shift = float(np.abs(
np.asarray(corrected[dependent_variable], dtype=float)
- np.asarray(score_data_df[dependent_variable], dtype=float)
).mean()) if len(score_data_df) else 0.0
n_batches = len(getattr(correction_report, 'batches', ()) or ())
print(
f"Batch correction {correction_report.method}: "
f"{correction_report.centroid_spread_before} -> "
f"{correction_report.centroid_spread_after} centroid spread, "
f"across {n_batches} batch(es); "
f"{dependent_variable} moved by {shift:.6g} on average. "
f"Report: {report_path}"
)
if n_batches < 2:
print(
f" It changed nothing, and could not: batch correction "
f"removes variance BETWEEN batches and this run has "
f"{n_batches}. Set batch_column to a column that varies, or "
f"batch_correction='none' -- the result is identical either "
f"way."
)
elif shift == 0.0:
print(
" It changed nothing: the batches already agree on "
f"{dependent_variable}. That is a finding about the screen, "
"not a failure of the correction."
)
score_data_df.loc[:, dependent_variable] = corrected[
dependent_variable
]
for note in correction_report.warnings:
print(f"Warning: batch correction: {note}")
save_settings(settings, name='regression', show=True)
count_source = os.path.dirname(settings['count_data'][0])
volcano_path = os.path.join(res_folder, 'volcano_plot.pdf')
if isinstance(settings['filter_value'], list):
filter_value = list(settings['filter_value'])
else:
filter_value = []
try:
from .well_spec import control_block_wells
for well in control_block_wells(settings):
if well not in filter_value:
filter_value.append(well)
except Exception: # noqa: BLE001
LOG.debug("could not resolve the control blocks", exc_info=True)
filter_column = settings['filter_column']
score_data_df = clean_controls(score_data_df, settings['filter_value'], filter_column)
try:
from .outlier_filter import apply as _drop_outliers, describe
score_data_df, _outlier_report = _drop_outliers(score_data_df,
settings)
_said = describe(_outlier_report)
if _said:
print(_said)
except Exception as _error: # noqa: BLE001
print(f"[outliers] the pre-annotation filter did not run "
f"({type(_error).__name__}: {_error}); the counts below are "
f"unfiltered")
if settings['verbose']:
print(f"Dependent variable after clean_controls: {len(score_data_df)}")
_AUTOMATIC_SETTINGS.clear()
screen_folders = _screen_figure_folders(settings)
sim_min_count = minimum_cell_simulation(
settings, tolerance=settings['tolerance'], dst=res_folder)
if settings['min_cells_per_well'] is None:
settings['min_cells_per_well'] = sim_min_count
_AUTOMATIC_SETTINGS['min_cells_per_well'] = sim_min_count
if settings['verbose']:
print(f"Minimum cell count: {settings['min_cells_per_well']}")
print(f"Dependent variable after minimum cell count filter: {len(score_data_df)}")
display(score_data_df)
orig_dv = settings['dependent_variable']
_before_transform = None
try:
_before_transform, _ = process_scores(
score_data_df, settings['dependent_variable'], None,
settings['min_cells_per_well'], settings['agg_type'],
None, settings['regression_type'],
settings['invert_dependent_variable'])
except Exception: # noqa: BLE001
_before_transform = None
dependent_df, dependent_variable = process_scores(
score_data_df, settings['dependent_variable'], None,
settings['min_cells_per_well'], settings['agg_type'],
settings['transform'], settings['regression_type'],
settings['invert_dependent_variable'])
_show_response_distribution(_before_transform, dependent_variable,
settings)
if settings['verbose']:
print(f"Dependent variable after process_scores: {len(dependent_df)}")
display(dependent_df)
if settings.get('calibrate_fraction_threshold'):
measured = _calibrated_fraction_threshold(settings)
if measured is not None:
settings['fraction_threshold'] = measured
_AUTOMATIC_SETTINGS['fraction_threshold'] = measured
if settings['fraction_threshold'] is None:
before_sweep = _figure_stamps(screen_folders)
settings['fraction_threshold'] = _graph_sequencing_stats(settings)
_AUTOMATIC_SETTINGS['fraction_threshold'] = settings['fraction_threshold']
for kept in _keep_figures_with_the_run(before_sweep, screen_folders,
res_folder):
print(f"Kept with the run: {kept}")
else:
before_sweep = _figure_stamps(screen_folders)
_draw_the_threshold_sweep(
settings, res_folder,
measured='fraction_threshold' in _AUTOMATIC_SETTINGS)
for kept in _keep_figures_with_the_run(before_sweep, screen_folders,
res_folder):
print(f"Kept with the run: {kept}")
if _AUTOMATIC_SETTINGS:
print("\nChosen automatically (not set by the user):")
for key, value in _AUTOMATIC_SETTINGS.items():
print(f" {key:<28}{value}")
try:
save_settings(settings, name='regression', show=False)
except Exception as error: # noqa: BLE001
print(f"Could not re-save the resolved settings: {error}")
_exclusions = settings.setdefault("_regression_exclusions", {})
_stage(settings, "reading the counts")
_read_kwargs = {
"filter_column": filter_column,
"filter_value": filter_value,
"record": _exclusions,
}
if settings.get('exclude_grnas'):
_read_kwargs["exclude_grnas"] = settings['exclude_grnas']
independent_df = process_reads(
count_data_df, settings['fraction_threshold'], None,
**_read_kwargs)
if settings['verbose']:
print("independent_df columns:", list(independent_df.columns))
print("independent_df head:")
print(independent_df.head())
print(independent_df)
if settings['verbose']:
print(f"Independent variable after process_reads: {len(independent_df)}")
merge_validate = (
'many_to_many' if settings['agg_type'] is None else 'many_to_one')
merged_df = pd.merge(independent_df, dependent_df, on='prc',
validate=merge_validate)
_check_score_count_pairing(independent_df, dependent_df, merged_df,
record=settings.get('_regression_exclusions'))
_merged_for_counts, n_grna, n_gene = _count_variable_instances(
merged_df, column_1='grna', column_2='gene')
if settings['verbose']:
display(independent_df)
display(dependent_df)
display(merged_df)
_assign_prc_parts(merged_df)
try:
os.makedirs(res_folder, exist_ok=True)
data_path = os.path.join(res_folder, 'regression_data.csv')
merged_df.to_csv(data_path, index=False)
print(f"Saved regression data to {data_path}")
qc_graph_type = _qc_graph_type()
cell_settings = {'src':data_path,
'graph_name':'cell_count',
'data_column':['cell_count'],
'grouping_column':'plateID',
'graph_type':qc_graph_type,
'theme':'bright',
'save':True,
'y_lim':[None,None],
'log_y':False,
'log_x':False,
'representation':'well',
'remove_outliers':False,
'verbose':False}
_, _ = _qc_plot(cell_settings)
final_grna_df, prc_gene_count_df = grna_metricks(merged_df)
if settings['outlier_detection']:
outliers_grna = get_outlier_reference_values(final_grna_df,outlier_col='grna_well_count',return_col='grna')
if len (outliers_grna) > 0:
merged_df = merged_df[
~merged_df['grna'].isin(outliers_grna)].copy()
final_grna_df, prc_gene_count_df = grna_metricks(merged_df)
merged_df.to_csv(data_path, index=False)
print(f"Saved regression data to {data_path}")
grna_data_path = os.path.join(res_folder, 'grna_well.csv')
final_grna_df.to_csv(grna_data_path, index=False)
print(f"Saved grna per well data to {grna_data_path}")
wells_per_gene_settings = {'src':grna_data_path,
'graph_name':'wells_per_gene',
'data_column':['grna_well_count'],
'grouping_column':'plateID',
'graph_type':qc_graph_type,
'theme':'bright',
'save':True,
'y_lim':[None,None],
'log_y':False,
'log_x':False,
'representation':'object',
'remove_outliers':False,
'verbose':True}
_, _ = _qc_plot(wells_per_gene_settings)
grna_well_data_path = os.path.join(res_folder, 'well_grna.csv')
prc_gene_count_df.to_csv(grna_well_data_path, index=False)
print(f"Saved well per grna data to {grna_well_data_path}")
grna_per_well_settings = {'src':grna_well_data_path,
'graph_name':'gene_per_well',
'data_column':['gene_count'],
'grouping_column':'plateID',
'graph_type':qc_graph_type,
'theme':'bright',
'save':True,
'y_lim':[None,None],
'log_y':False,
'log_x':False,
'representation':'well',
'remove_outliers':False,
'verbose':False}
_, _ = _qc_plot(grna_per_well_settings)
except Exception as e:
print(e)
if str(settings.get('inference', 'parametric')).lower() == 'auto':
resolved_mode, reason = resolve_auto_inference(merged_df, settings)
settings['analysis_mode'] = resolved_mode
print(f"inference='auto': {reason}")
elif settings.get('analysis_mode') == 'regression':
for _one in resolve_levels(settings.get('regression_type'),
settings.get('level', 'both')):
warning = _identifiability_warning(merged_df, settings,
level=_one)
if warning:
print(f" level={_one!r}:")
print(warning)
if settings.get('analysis_mode') == 'guide_permutation':
_chosen = settings.get('regression_type')
if _chosen:
print(f"inference='nonparametric': this is a permutation test, so "
f"it fits no model and regression_type={_chosen!r} is not "
f"read. Choosing a different regression_type with this "
f"inference gives the same numbers; set "
f"inference='parametric' to fit {_chosen!r} itself.")
_stage(settings, "permuting the guides")
output = _run_guide_permutation_analysis(
merged_df, dependent_variable, res_folder, settings)
_stage(settings, "the permutation has returned")
if settings.get('verbose'):
print(
f"Guide permutation analysis tested "
f"{len(output['primary'])} guides in the primary "
f">={output['primary_min_wells']}-well family and called "
f"{len(output['significant'])} at "
f"{settings['multiple_testing_method']} "
f"alpha={settings['fdr_alpha']}."
)
try:
from .regression_summary import write_run_summary
except ImportError:
pass
else:
try:
write_run_summary(
res_folder, model=None, settings=settings,
coef_df=output.get('primary'),
regression_type=settings.get('regression_type'),
fit_designs={})
except Exception as error: # noqa: BLE001 - never lose a run
print(f"Could not write the run summary: "
f"{type(error).__name__}: {error}")
output.setdefault('res_folder', res_folder)
output.setdefault('settings', dict(settings))
output.setdefault('regression_type', settings.get('regression_type'))
return output
if not _show_plates(merged_df, orig_dv, res_folder):
_ = plot_plates(merged_df, variable=orig_dv, grouping='mean',
min_max='allq', cmap='viridis', min_count=None,
dst=res_folder)
_stage(settings, "fitting the model")
fits = regression_levels(
merged_df, csv_path, dependent_variable=dependent_variable,
regression_type=settings['regression_type'],
regression_backend=settings.get('regression_backend',
DEFAULT_REGRESSION_BACKEND),
level=settings.get('level', 'both'),
alpha=settings['alpha'],
random_row_column_effects=settings['random_row_column_effects'],
model_plate_position=settings.get('model_plate_position', True),
model_data_layout=settings.get('model_data_layout', 'long'),
nc=settings['negative_control_id'], pc=settings['positive_control_id'],
controls=settings['nontargeting_control_grnas'], dst=res_folder,
verbose=bool(settings.get('verbose')),
transform=str(settings.get('transform') or ''),
cov_type=settings['cov_type'],
l1_ratio=settings['l1_ratio'],
quantile=settings['quantile'],
hinge_threshold=settings['hinge_threshold'],
hinge_n_boot=settings['hinge_n_boot'],
huber_t=settings['huber_t'],
spline_knots=settings.get('spline_knots', 4),
spline_degree=settings.get('spline_degree', 3),
group_lasso_lambda=settings.get('group_lasso_lambda', 'auto'),
rra_alpha=settings.get('rra_alpha', 0.25),
rra_permutations=settings.get('rra_permutations', 10000),
qc=bool(settings.get('regression_qc', True)),
legacy_volcano=bool(settings.get('legacy_volcano', False)),
intercept=str(settings.get('intercept') or 'fitted'),
intercept_value=float(settings.get('intercept_value') or 0.0),
)
regression_type = next(iter(fits.values()))[2]
fit_designs = {one: dict(one_coef.attrs.get('fit_design', {}))
for one, (_model, one_coef, _type) in fits.items()}
settings['_regression_diagnostics'] = _write_regression_diagnostics(
res_folder, merged_df, fits, settings)
level_tables = {
one: _annotate_level_coefficients(one_coef, n_grna, n_gene)
for one, (_model, one_coef, _type) in fits.items()
}
for table in level_tables.values():
table.attrs.pop('fit_design', None)
if regression_type == 'mixed' and 'gene' in level_tables:
whole = level_tables.pop('gene')
blups = whole['term_type'] == TERM_BLUP
level_tables['gene'] = whole.loc[~blups].copy()
guide_table = whole.loc[blups].copy()
guide_table['level'] = 'grna'
guide_table['q_value'] = np.nan
guide_table['multiple_testing_method'] = 'none'
level_tables['grna'] = guide_table
print(f"Mixed fit: {len(level_tables['gene'])} gene rows corrected as "
f"one family, {len(guide_table)} guide BLUPs written without a "
f"q value. Choose a fixed-effects model with level='grna' for a "
f"guide-level hit list.")
corrected = {}
hits_by_level = {}
thresholds_by_level = {}
for one, table in level_tables.items():
if regression_type == 'mixed' and one == 'grna':
corrected[one] = table
hits_by_level[one] = table.iloc[0:0]
thresholds_by_level[one] = 0
continue
table, level_hits, level_threshold, _rule = _call_level_hits(
table, one, settings, regression_type, merged_df,
dependent_variable, bootstrap=bootstrap_selection_frequencies)
corrected[one] = table
hits_by_level[one] = level_hits
thresholds_by_level[one] = level_threshold
primary = 'grna' if 'grna' in fits else next(iter(fits))
model = fits[primary][0]
reg_threshold = thresholds_by_level.get(primary, 0)
grna_coef_df = corrected.get('grna')
gene_coef_df = corrected.get('gene')
if grna_coef_df is not None:
grna_coef_df = grna_coef_df.dropna(subset=['n_grna'])
if gene_coef_df is not None:
gene_coef_df = gene_coef_df.dropna(subset=['n_gene'])
template = corrected[primary].iloc[0:0]
if grna_coef_df is None:
print("level='gene': no guide fit was run, so results_grna.csv is "
"written empty. Set level='both' or level='grna' for one.")
grna_coef_df = template
if gene_coef_df is None:
print("level='grna': no gene fit was run, so results_gene.csv is "
"written empty. Set level='both' or level='gene' for one.")
gene_coef_df = template
def _stack(frames):
"""Concatenate nonempty frames, or return the empty result template."""
kept = [frame for frame in frames if len(frame)]
return pd.concat(kept, ignore_index=True) if kept else template
coef_df = _stack(corrected.values())
significant = _stack(hits_by_level.values())
if _annotation_source(settings):
from .annotation import annotate_with, supplementary
source = _annotation_source(settings)
cache = _annotation_cache(settings)
annotated, notes = {}, []
for name, frame in (('results', coef_df), ('gene', gene_coef_df),
('grna', grna_coef_df),
('significant', significant)):
before = len(frame)
annotated[name], note = annotate_with(
frame, source, cache_dir=cache, quiet=(name != 'results'))
if note and name == 'results':
notes.append(note)
if len(annotated[name]) != before:
raise ValueError(
f"the {source} annotation changed {name} from {before} "
f"to {len(annotated[name])} row(s).")
for note in notes:
print(f"Annotation: {note}")
coef_df = annotated['results']
gene_coef_df = annotated['gene']
grna_coef_df = annotated['grna']
significant = annotated['significant']
supplementary(
coef_df['feature'] if 'feature' in coef_df.columns else None,
path=os.path.join(res_folder, 'supplementary_topology.csv'))
coef_df.to_csv(results_path, index=False)
gene_coef_df.to_csv(results_path_gene, index=False)
grna_coef_df.to_csv(results_path_grna, index=False)
if regression_type in ['ols', 'beta']:
if settings['verbose']:
print(model.summary())
save_summary_to_file(
model, file_path=os.path.join(res_folder, SUMMARY_FILENAME))
try:
from .regression_summary import write_run_summary
except ImportError:
pass
else:
try:
_stage(settings, "the fit has returned")
write_run_summary(res_folder, model=model, settings=settings,
coef_df=coef_df, regression_type=regression_type,
fit_designs=fit_designs)
except Exception as error: # noqa: BLE001 - never lose a run
print(f"Could not write the run summary: "
f"{type(error).__name__}: {error}")
significant.to_csv(hits_path, index=False)
threshold = settings['min_observations_per_hit']
significant_grna_filtered = significant[significant['n_grna'] > threshold]
significant_gene_filtered = significant[significant['n_gene'] > threshold]
significant_filtered = pd.concat([significant_grna_filtered, significant_gene_filtered])
filtered_hit_path = os.path.join(os.path.dirname(hits_path), 'results_significant_filtered.csv')
significant_filtered.to_csv(filtered_hit_path, index=False)
if isinstance(settings['metadata_files'], str):
settings['metadata_files'] = [settings['metadata_files']]
results_metadata_df = tabular.read_table(results_path, report=None)
gene_merged_df = tabular.read_table(results_path_gene, report=None)
grna_merged_df = tabular.read_table(results_path_grna, report=None)
for metadata_file in settings['metadata_files']:
file = os.path.basename(metadata_file)
filename, _ = os.path.splitext(file)
try:
if not os.path.isfile(metadata_file) \
or os.path.getsize(metadata_file) == 0:
print(f"Skipping empty or missing metadata file: "
f"{metadata_file}")
continue
except OSError:
continue
try:
_ = merge_regression_res_with_metadata(hits_path, metadata_file, name=filename)
results_metadata_df = merge_regression_res_with_metadata(results_path, metadata_file, name=filename)
gene_merged_df = merge_regression_res_with_metadata(results_path_gene, metadata_file, name=filename)
grna_merged_df = merge_regression_res_with_metadata(results_path_grna, metadata_file, name=filename)
except Exception as metadata_error:
print(f"Could not merge metadata from {metadata_file}: "
f"{metadata_error}")
continue
draw_legacy_volcano = bool(settings.get('legacy_volcano', False))
if not draw_legacy_volcano:
print("Legacy volcano: off (the interactive volcano and the house-"
"style figure are drawn instead). Set legacy_volcano=True to "
"draw the original matplotlib one as well.")
if _toxoplasma_is_on(settings):
data_path = results_metadata_df
data_path_gene = gene_merged_df
data_path_grna = grna_merged_df
base_dir = os.path.dirname(os.path.abspath(__file__))
metadata_path = os.path.join(base_dir, 'resources', 'data', 'lopit.csv')
gene_list = custom_volcano_plot(
gene_merged_df, metadata_path, metadata_column='tagm_location',
point_size=600, figsize=20, threshold=reg_threshold,
save_path=volcano_path, x_lim=settings.get('x_lim'),
y_lims=settings.get('y_lims'),
draw=draw_legacy_volcano,
)
if not draw_legacy_volcano:
pass
elif os.path.exists(volcano_path):
print(f"Saved volcano plot to {volcano_path}")
else:
print(f"WARNING: the legacy volcano was requested but no file was "
f"written to {volcano_path}")
display(gene_list) if gene_list is not None else None
phenotype_plot = os.path.join(res_folder, 'phenotype_plot.pdf')
transcription_heatmap = os.path.join(res_folder, 'transcription_heatmap.pdf')
metadata_files = list(settings.get('metadata_files') or [])
have_curated_tables = len(metadata_files) >= 2
if not have_curated_tables:
print(f"Skipping the phenotype and transcription reports: they "
f"need two curated metadata tables (GT1 phenotypes and "
f"ME49 expression) and {len(metadata_files)} were given. "
f"The volcano and every results table are unaffected.")
data_GT1 = (tabular.read_table(metadata_files[1], low_memory=False,
canonicalise=False, report=None)
if have_curated_tables else None)
data_ME49 = (tabular.read_table(metadata_files[0], low_memory=False,
canonicalise=False, report=None)
if have_curated_tables else None)
columns = ['sense - Tachyzoites', 'sense - Tissue cysts',
'sense - EES1', 'sense - EES2', 'sense - EES3',
'sense - EES4', 'sense - EES5']
if gene_list and have_curated_tables:
print('Plotting gene phenotypes and heatmaps')
print(gene_list)
plot_gene_phenotypes(data=data_GT1, gene_list=gene_list,
save_path=phenotype_plot)
plot_gene_heatmaps(
data=data_ME49, gene_list=gene_list, columns=columns,
x_column='Gene ID', normalize=True,
save_path=transcription_heatmap,
)
elif not gene_list:
print("No gene_list produced; skipping phenotype and heatmap plots.")
if not _toxoplasma_is_on(settings) and draw_legacy_volcano:
try:
from .plot import volcano_plot as _plain_volcano
_source = results_path_gene
_plain_volcano(
_source,
fold_change_col='coefficient',
p_value_col='p_value',
name_col='feature',
x_transform='none', y_transform='-log10',
fold_change_threshold=reg_threshold,
p_value_threshold=float(settings.get('fdr_alpha', 0.05) or 0.05),
point_size=20.0, figsize=(10.0, 8.0),
title=f"{settings.get('regression_type', 'ols')} - gene",
save_path=volcano_path, show=False)
except Exception as _volcano_error:
print(f"Could not draw the volcano plot: "
f"{type(_volcano_error).__name__}: {_volcano_error}")
if os.path.exists(volcano_path):
print(f"Saved volcano plot to {volcano_path}")
print('Significant Genes')
grnas = significant['grna'].unique().tolist()
genes = significant['gene'].unique().tolist()
print(f"Found p<0.05 coedfficients for {len(grnas)} gRNAs and {len(genes)} genes")
display(significant)
_warn_if_penalised_no_hits(settings, coef_df)
try:
from .guide_concordance import concordance_report
controls = {}
for _key, _role in (('positive_control_id', 'positive'),
('negative_control_id', 'negative')):
_value = settings.get(_key)
if _value not in (None, ''):
controls[str(_value)] = _role
print()
print(concordance_report(
coef_df, alpha=float(settings.get('fdr_alpha', 0.05) or 0.05),
controls=controls))
except Exception as concordance_error:
print(f"Could not summarise guide support: {concordance_error}")
output = {'results':coef_df,
'significant':significant,
'model': model,
'model_data': merged_df,
'fit_designs': fit_designs,
'regression_type': regression_type,
'res_folder': res_folder,
'settings': dict(settings)}
manifest = getattr(coef_df, "attrs", {}).get("qc_manifest")
if manifest:
output['qc'] = manifest
worst = manifest.get('verdict')
if worst is not None:
output['qc_verdict'] = worst
output['qc_verdict_level'] = manifest.get('verdict_level', 'unknown')
return output
#: The fixed head of a ``prcfo`` key, in order. The object id is always the
#: LAST token and anything between the two is the timepoint, which is how a
#: five-token and a six-token key are told apart without guessing.
_PRCFO_HEAD = schema.FIELD_KEY_COLUMNS
def _assign_prcfo_parts(df, object_column='objectID'):
"""Split ``prcfo`` into its named components and assign them onto ``df``.
``prcfo`` is written by :func:`spacr.utils._map_wells_png` and rebuilt by
:func:`spacr.utils._split_data`. It has **five** tokens on a plain screen
(``plate_row_column_field_object``) and **six** on a timelapse
(``plate_row_column_field_TIME_object``).
Three places in this module used to spell that as
.. code-block:: python
df[['plateID', 'rowID', 'columnID', 'fieldID', 'objectID']] = \\
df['prcfo'].str.split('_', expand=True)
which is not a mis-assignment on a timelapse — it is a hard stop. Six
split columns against five keys makes pandas raise ``ValueError: Columns
must be same length as key``, so :func:`ml_analysis` threw away a
completed model at its very last statement (measured on a real 2-well x
2-field x 3-frame x 3-object database: 36 rows in, fit and permutation
importance done, then ``ValueError`` at ``ml.py:2517``). The five names
would *also* have been wrong had it not raised — the fifth token of a
timelapse key is the timepoint, so ``objectID`` would have held ``'t1'``
and the object id would have been dropped entirely.
Splitting the head from the left and the object from the right recovers
both forms, and the timepoint is kept rather than discarded: it is written
under whichever spelling ``df`` already uses (``timeID`` canonical,
``time_id`` legacy — resolved through :func:`spacr.utils._time_column`),
defaulting to ``timeID``.
This doubles as repair-on-read for a scores CSV whose ``objectID`` was
filled in by a positional guess over a timelapse crop name — the same
guess :func:`spacr.ml.interperate_vision_model` already refuses to trust —
because the components are recomputed from ``prcfo`` and overwrite what is
there.
:param df: Frame carrying a ``prcfo`` column.
:param object_column: Name to give the object id. ``'objectID'`` for the
read/score paths, ``'object'`` in :func:`ml_analysis`, which is what
each of them already wrote.
:returns: ``df``, with the component columns assigned.
:raises TimelapseKeyMismatch: when the frame mixes five- and six-token
keys — two runs that disagreed about ``timelapse`` were concatenated,
and there is no single answer to what the fifth token means.
:raises ValueError: when a key has neither five nor six tokens.
"""
from .io import TimelapseKeyMismatch
from .utils import _time_column
values = df['prcfo']
tokens = values.astype(str).str.split(schema.KEY_SEPARATOR)
widths = tokens.map(len)
seen = set(widths.unique().tolist())
unexpected = sorted(seen - {5, 6})
if unexpected:
example = values[widths.isin(unexpected)].iloc[0]
raise ValueError(
f"prcfo must be plate_row_column_field_object (5 tokens) or "
f"plate_row_column_field_time_object (6, timelapse); found "
f"{unexpected} token(s), e.g. {example!r}."
)
if seen == {5, 6}:
raise TimelapseKeyMismatch(
f"prcfo mixes {int((widths == 5).sum())} key(s) without a "
f"timepoint and {int((widths == 6).sum())} with one, so the fifth "
f"token is an object id in some rows and a timepoint in others. "
f"Two runs that disagreed about 'timelapse' have been combined; "
f"re-run the non-timelapse half rather than splitting this."
)
parsed = [schema.parse_prcfo(value) for value in values.astype(str)]
for name in _PRCFO_HEAD:
df[name] = [getattr(obj, name) for obj in parsed]
df[object_column] = [obj.objectID for obj in parsed]
if seen == {6}:
df[_time_column(df.columns) or schema.TIME_KEY] = [
obj.timeID for obj in parsed]
return df
[docs]
def process_reads(csv_path, fraction_threshold, plate, filter_column=None,
filter_value=None, record=None, exclude_grnas=None):
"""Load a per-gRNA read-count CSV and return per-well normalised fractions.
Splits derived ``plate_row`` or ``prcfo`` identifiers, computes each
gRNA's fraction of the well total, applies an optional
fraction-cutoff filter and returns a compact ``(prc, grna, fraction)``
frame (with ``gene`` derived from the gRNA when possible).
:param csv_path: Path to the counts CSV, or an already-loaded DataFrame.
:param fraction_threshold: Drop rows below this fraction; must be in
``[0, 1]`` or ``None``.
:param plate: Plate identifier used when no ``plateID`` column is
present.
:param filter_column: Column (or list of columns) to filter rows on.
:param filter_value: Values (or list of values) to drop from
``filter_column``.
:param record: Optional mutable mapping that records exclusions for the
persisted regression summary.
:param exclude_grnas: Guide or gene identifiers to remove from the raw
count table. Gene identifiers remove all associated guides. This is
applied before well totals and fractions are calculated, so retained
guides are normalised against the retained read count.
:returns: DataFrame with columns ``prc``, ``grna``, ``fraction``.
:raises ValueError: on missing required columns, invalid
``fraction_threshold``, or when the threshold removes all rows.
"""
from .utils import correct_metadata
if isinstance(csv_path, pd.DataFrame):
csv_df = csv_path
else:
csv_df = tabular.read_table(csv_path)
csv_df = correct_metadata(csv_df)
if 'grna_name' in csv_df.columns:
csv_df = csv_df.rename(columns={'grna_name': 'grna'})
if exclude_grnas and 'grna' in csv_df.columns:
from .read_background import resolve_exclusions, unmatched_exclusions
requested = ([exclude_grnas] if isinstance(exclude_grnas, str)
else list(exclude_grnas))
guide_names = csv_df['grna'].astype(str)
gene_names = (csv_df['gene'].astype(str)
if 'gene' in csv_df.columns else None)
resolved = resolve_exclusions(requested, guide_names, gene_names)
unmatched = unmatched_exclusions(requested, guide_names, gene_names)
drop_mask = guide_names.isin(resolved)
rows_before = len(csv_df)
rows_removed = int(drop_mask.sum())
if rows_removed:
csv_df = csv_df.loc[~drop_mask].copy()
print(
f"Excluded {rows_removed} of {rows_before} raw count rows "
f"spanning {len(resolved)} guide(s) named by exclude_grnas "
f"before well totals and fractions were calculated: "
f"{', '.join(sorted(resolved)[:5])}"
f"{' ...' if len(resolved) > 5 else ''}."
)
if unmatched:
print(
f"exclude_grnas named {len(unmatched)} value(s) that match "
f"no guide or gene in the raw count table: "
f"{', '.join(map(str, unmatched[:5]))}"
f"{' ...' if len(unmatched) > 5 else ''}."
)
if record is not None:
record["exclude_grnas"] = (
record.get("exclude_grnas", 0) + rows_removed)
record["exclude_grnas_of"] = (
record.get("exclude_grnas_of", 0) + rows_before)
prior_guides = record.get("exclude_grnas_guides", ())
record["exclude_grnas_guides"] = sorted(
set(map(str, prior_guides)) | set(map(str, resolved)))
prior_unmatched = record.get("exclude_grnas_unmatched", ())
record["exclude_grnas_unmatched"] = sorted(
set(map(str, prior_unmatched)) | set(map(str, unmatched)))
if csv_df.empty:
raise ValueError(
f"exclude_grnas removed all {rows_before} raw count rows. "
"Remove or narrow the exclusion before running regression."
)
if 'plate_row' in csv_df.columns:
pieces = csv_df['plate_row'].astype(str).str.rsplit(
schema.KEY_SEPARATOR, n=1)
malformed = pieces.map(len) < 2
if malformed.any():
example = csv_df.loc[malformed, 'plate_row'].iloc[0]
raise ValueError(
f"'plate_row' must be '<plate>{schema.KEY_SEPARATOR}<row>', "
f"but {int(malformed.sum())} of {len(csv_df)} value(s) hold no "
f"{schema.KEY_SEPARATOR!r}, e.g. {example!r}. Supply separate "
f"'plateID' and 'rowID' columns instead, or repair the count "
f"table — guessing which half is the plate would key the whole "
f"screen on the wrong well.")
csv_df['plateID'] = pieces.str[0]
csv_df['rowID'] = pieces.str[-1]
if not 'plateID' in csv_df.columns:
if not plate is None:
csv_df['plateID'] = plate
else:
csv_df['plateID'] = 'plate1'
if 'prcfo' in csv_df.columns:
csv_df = _assign_prcfo_parts(csv_df, object_column='objectID')
csv_df['prc'] = _compose_prc_column(csv_df)
if isinstance(filter_column, str):
filter_column = [filter_column]
if isinstance(filter_value, str):
filter_value = [filter_value]
if isinstance(filter_column, list):
for filter_col in filter_column:
for value in filter_value:
csv_df = csv_df.loc[csv_df[filter_col] != value].copy()
if not all(col in csv_df.columns for col in ['rowID','columnID','grna','count']):
raise ValueError("The CSV file must contain 'grna', 'count', 'rowID', and 'columnID' columns.")
csv_df['prc'] = _compose_prc_column(csv_df)
grouped_df = csv_df.groupby('prc')['count'].sum().reset_index()
grouped_df = grouped_df.rename(columns={'count': 'total_counts'})
merged_df = pd.merge(csv_df, grouped_df, on='prc', validate='many_to_one')
merged_df['fraction'] = merged_df['count'] / merged_df['total_counts']
if fraction_threshold is not None:
if not 0 <= fraction_threshold <= 1:
raise ValueError(
f"fraction_threshold={fraction_threshold} is outside the valid range [0, 1]. "
f"The 'fraction' column is a relative abundance bounded between 0 and 1."
)
observations_before = len(merged_df)
frac_min = merged_df['fraction'].min()
frac_max = merged_df['fraction'].max()
frac_median = merged_df['fraction'].median()
merged_df = merged_df[merged_df['fraction'] >= fraction_threshold]
observations_after = len(merged_df)
removed = observations_before - observations_after
if record is not None:
record["fraction_threshold"] = (
record.get("fraction_threshold", 0) + int(removed))
record["fraction_threshold_of"] = (
record.get("fraction_threshold_of", 0) + int(observations_before))
pct_retained = 100 * observations_after / observations_before if observations_before else 0
print(
f"Removed {removed} of {observations_before} observations "
f"below fraction threshold {fraction_threshold} "
f"({pct_retained:.1f}% retained). "
f"Fraction range in input: [{frac_min:.4g}, {frac_max:.4g}], median {frac_median:.4g}."
)
if observations_after == 0:
raise ValueError(
f"All {observations_before} rows were removed by fraction_threshold={fraction_threshold}. "
f"Observed fraction range was [{frac_min:.4g}, {frac_max:.4g}], median {frac_median:.4g}. "
f"Choose a threshold below the median, or pass None to auto-compute."
)
merged_df = merged_df[['prc', 'grna', 'fraction']]
tokens = merged_df['grna'].astype(str).str.split(schema.KEY_SEPARATOR)
widths = sorted(set(tokens.map(len).tolist()))
if widths == [3]:
merged_df['gene'] = tokens.str[1]
merged_df['grna'] = (tokens.str[1] + schema.KEY_SEPARATOR
+ tokens.str[2])
else:
example = merged_df['grna'].iloc[0] if len(merged_df) else None
print(f"Not splitting 'grna' into org/gene/grna: that split is "
f"positional and needs every name to be "
f"'<org>{schema.KEY_SEPARATOR}<gene>{schema.KEY_SEPARATOR}"
f"<guide>' (3 components), but this table holds names with "
f"{widths} component(s), e.g. {example!r}. No 'gene' column "
f"is produced; a step that needs one will name it.")
return merged_df
#: The squeeze applied before a logit, and the reason it exists.
#:
#: A classification score is a PROPORTION, and a screen produces exact 0 and
#: exact 1 -- neither of which has a logit. Smithson and Verkuilen's transform
#: pulls the whole scale off the endpoints by (n-1)/n plus a half, which is
#: the standard treatment and is reported in the run summary rather than
#: applied quietly: a transform that silently moved a user's 0 to 0.001
#: changed their data.
BETA_SQUEEZE_NOTE = (
"beta: the response was mapped to the logit scale. A proportion of "
"exactly 0 or 1 has no logit, so the scale was squeezed off its "
"endpoints by the Smithson-Verkuilen rule ((y*(n-1)+0.5)/n) first")
[docs]
def beta_logit(values):
"""A proportion on the logit scale, with the endpoints squeezed in.
``transform='beta'`` is intended for proportional responses such as
classification scores and their well aggregates, where a logarithm is
not appropriate.
This is distinct from ``regression_type='beta'``, which selects a beta
GLM. One transforms the response; the other selects the model family.
:param values: proportions in ``[0, 1]``, array-like; converted to a
float array. Non-finite entries pass through unchanged. When any
finite value is at or beyond 0 or 1 the finite values are squeezed
with ``(y * (n - 1) + 0.5) / n`` before the logit.
"""
array = np.asarray(values, dtype=float)
finite = np.isfinite(array)
n = int(finite.sum())
if n < 1:
return array
squeezed = array.copy()
inside = array[finite]
if inside.min() <= 0.0 or inside.max() >= 1.0:
squeezed[finite] = (inside * (n - 1) + 0.5) / n
squeezed[finite] = np.clip(squeezed[finite], 1e-9, 1.0 - 1e-9)
out = np.array(array, dtype=float, copy=True)
out[finite] = np.log(squeezed[finite] / (1.0 - squeezed[finite]))
return out
[docs]
def check_normality(data, variable_name, verbose=False):
"""Check if the data is normally distributed using the Shapiro-Wilk test.
:param data: numeric values, array-like; non-finite values are dropped
and fewer than 3 remaining values returns ``False`` without testing.
:param variable_name: name printed in the verbose messages only.
:param verbose: print the test statistic, P value and verdict.
:returns: ``True`` when the Shapiro-Wilk P value exceeds 0.05.
"""
values = np.asarray(data, dtype=float)
values = values[np.isfinite(values)]
if values.size < 3:
if verbose:
print(f"Shapiro-Wilk Test for {variable_name}: at least 3 finite "
f"values are required; received {values.size}.")
return False
stat, p_value = shapiro(values)
if verbose:
print(f"Shapiro-Wilk Test for {variable_name}:\nStatistic: {stat}, P-value: {p_value}")
if p_value > 0.05:
if verbose:
print(f"Normal distribution: The data for {variable_name} is normally distributed.")
return True
else:
if verbose:
print(f"Normal distribution: The data for {variable_name} is not normally distributed.")
return False
[docs]
def clean_controls(df,values, column):
"""Drop rows whose ``column`` holds one of the listed ``values``.
:param df: Source DataFrame.
:param values: List of values to remove. Anything that is not a list
(a bare value included) is a no-op.
:param column: Column, or list of columns, to check. ``None`` is a
no-op.
:returns: Filtered DataFrame (unchanged if ``column`` is missing or
``values`` is not a list).
"""
if column is None:
return df
columns = list(column) if isinstance(column, (list, tuple, set)) else [column]
if isinstance(values, list):
for col in columns:
if col in df.columns:
for value in values:
df = df[~df[col].isin([value])]
print(f'Removed data from {value}')
return df
[docs]
def process_scores(df, dependent_variable, plate, min_cells_per_well=25, agg_type='mean', transform=None, regression_type='ols', invert_dependent_variable=False):
"""Aggregate per-object model scores to per-well summaries, ready for regression.
Ensures ``plateID/rowID/columnID/prc`` columns exist, applies an
optional inversion of the raw response, aggregates by well according
to ``agg_type`` (or with ``sum`` for the count models
``'poisson'`` and ``'horseshoe'``), enforces
``min_cells_per_well`` and optionally transforms the aggregated response.
:param df: Per-object score DataFrame.
:param dependent_variable: Column being aggregated.
:param plate: Plate identifier to stamp when the frame is
single-plate; ignored (with warning) when multiple plates exist.
:param min_cells_per_well: Wells with fewer objects are dropped.
Default ``25``.
:param agg_type: ``'mean'``, ``'median'``, ``'quantile'`` or None.
:param transform: Optional post-aggregation transform name
(see :func:`apply_transformation`).
:param regression_type: If ``'poisson'`` or ``'horseshoe'``, aggregation
uses ``sum`` - both model a per-well count, not a per-well average.
:param invert_dependent_variable: ``False``/``0`` = no inversion;
``True``/``1`` = ``1 - x``; ``-1`` = ``1 / x``.
:returns: ``(dependent_df, dependent_variable)`` — the per-well
DataFrame and the (possibly transformed) response column name.
:raises ValueError: on missing identifiers, unsupported ``agg_type``
or unrecognised ``invert_dependent_variable``.
"""
from .utils import correct_metadata
df = df.reset_index(drop=True)
if 'prcfo' in df.columns:
df = df.loc[:, ~df.columns.duplicated()].copy()
if not all(col in df.columns for col in ['plateID', 'rowID', 'columnID']):
df = _assign_prcfo_parts(df, object_column='objectID')
df['prc'] = _compose_prc_column(df)
else:
df = correct_metadata(df)
df = df.loc[:, ~df.columns.duplicated()].copy()
n_plates_in_df = df['plateID'].nunique(dropna=True) if 'plateID' in df.columns else 0
if plate is not None:
if n_plates_in_df > 1:
print(f"Warning: process_scores received plate={plate!r} but the input "
f"DataFrame already contains {n_plates_in_df} distinct plateIDs. "
f"Ignoring the 'plate' argument and using the per-row plateID "
f"column to avoid collapsing plates.")
else:
df['plateID'] = plate
if 'plateID' not in df.columns or df['plateID'].isna().all():
raise ValueError(
"process_scores: DataFrame has no usable 'plateID' column "
"and no 'plate' argument was provided."
)
if all(col in df.columns for col in ['plateID', 'rowID', 'columnID']):
df['prc'] = _compose_prc_column(df)
else:
raise ValueError("The DataFrame must contain 'plateID', 'rowID', and 'columnID' columns.")
df = df[['prc', dependent_variable]]
df = df[['prc', dependent_variable]].copy()
if invert_dependent_variable in (True, 1):
df[dependent_variable] = 1.0 - df[dependent_variable]
print(f"Inverted '{dependent_variable}' as 1 - x on raw values.")
elif invert_dependent_variable == -1:
raw = df[dependent_variable]
n_zero = int((raw == 0).sum())
if n_zero > 0:
print(f"Warning: '{dependent_variable}' contains {n_zero} zero "
f"values; 1/x is undefined for those rows. They will be set "
f"to NaN and dropped from this analysis.")
df[dependent_variable] = 1.0 / raw.where(raw != 0)
df = df.dropna(subset=[dependent_variable])
print(f"Inverted '{dependent_variable}' as 1/x on raw values.")
elif invert_dependent_variable in (False, 0):
pass
else:
raise ValueError(
f"invert_dependent_variable must be one of False, True, 1, -1; "
f"got {invert_dependent_variable!r}."
)
grouped = df.groupby('prc')[dependent_variable]
count_models = ('poisson', 'horseshoe')
if regression_type not in count_models:
print(f'Using agg_type: {agg_type}')
if agg_type == 'median':
dependent_df = grouped.median().reset_index()
elif agg_type == 'mean':
dependent_df = grouped.mean().reset_index()
elif agg_type == 'quantile':
dependent_df = grouped.quantile(0.75).reset_index()
elif agg_type is None:
dependent_df = df.reset_index()
if 'prcfo' in dependent_df.columns:
dependent_df = dependent_df.drop(columns=['prcfo'])
else:
raise ValueError(f"Unsupported aggregation type {agg_type}")
if regression_type in count_models:
agg_type = 'count'
print(f'Using agg_type: {agg_type} for {regression_type} regression')
dependent_df = grouped.sum().reset_index()
summed = pd.to_numeric(dependent_df.get(dependent_variable),
errors='coerce')
if summed is not None and len(summed):
finite = summed[np.isfinite(summed)]
if len(finite) and not np.all(
np.isclose(finite, np.rint(finite), rtol=0, atol=1e-8)):
example = float(finite.iloc[0])
raise ValueError(
f"regression_type={regression_type!r} models the well's "
f"positive COUNT -- the number of cells called positive "
f"-- and gets it by summing {dependent_variable!r} per "
f"well. That column holds continuous scores, so the sum "
f"is {example:.4g} rather than a whole number of cells. "
f"Either fit a continuous model ('ols', 'mixed', or "
f"'beta', which is built for a proportion), or give "
f"dependent_variable a per-cell 0/1 label so its "
f"per-well sum is a real count.")
cell_count = grouped.size().reset_index(name='cell_count')
if agg_type is None:
dependent_df = pd.merge(dependent_df, cell_count, on='prc',
validate='many_to_one')
else:
dependent_df['cell_count'] = cell_count['cell_count']
print("1 test")
display(dependent_df)
dependent_df = dependent_df[dependent_df['cell_count'] >= min_cells_per_well]
print("2 test")
display(dependent_df)
is_normal = check_normality(dependent_df[dependent_variable], dependent_variable)
if transform is not None and regression_type in count_models:
print(f"Ignoring transform={transform!r}: {regression_type} models a "
f"per-well count, and a transformed count is not a count.")
transform = None
if transform == 'beta':
column = pd.to_numeric(dependent_df[dependent_variable],
errors='coerce')
inside = column[np.isfinite(column)]
if len(inside) and (inside.min() < 0.0 or inside.max() > 1.0):
raise ValueError(
f"transform='beta' puts the response on the logit scale, "
f"which is only defined for a proportion, but "
f"{dependent_variable!r} runs from {inside.min():.4g} to "
f"{inside.max():.4g}. Use a score or a fraction here, or "
f"pick transform='log' for a response in measured units.")
print(BETA_SQUEEZE_NOTE)
if transform is not None:
transformer = apply_transformation(dependent_df[dependent_variable], transform=transform)
transformed_var = f'{transform}_{dependent_variable}'
dependent_df[transformed_var] = transformer.fit_transform(dependent_df[[dependent_variable]])
dependent_variable = transformed_var
is_normal = check_normality(dependent_df[transformed_var], transformed_var)
if not is_normal:
print(f'{dependent_variable} is not normally distributed')
else:
print(f'{dependent_variable} is normally distributed')
return dependent_df, dependent_variable
@single_threaded_openmp('classical ML training')
@_flowview_pipeline("ml")
[docs]
def generate_ml_scores(settings):
"""Train a classical ML classifier (XGBoost / logistic / RF) on per-object features and score every well of a screen.
Reads the measurement store selected by ``measurement_backend`` from
:func:`spacr.measure.measure_crop`, merges cell/nucleus/pathogen/
cytoplasm feature tables, uses the wells marked as
``positive_control`` / ``negative_control`` (or an annotation column)
as training labels, delegates fitting to :func:`ml_analysis`, and
writes per-object predictions, permutation and feature-importance
tables plus a plate heatmap into ``results/`` under the first source folder.
:param settings: Settings dict, canonicalized via
:func:`spacr.settings.set_default_analyze_screen`. Key entries:
- ``src`` (str or list) — folder(s) containing the measurements.
- ``measurement_backend`` / ``measurement_backend_target`` — select
the SQLite, DuckDB, Parquet or PostgreSQL measurement store.
- ``channel_of_interest`` — 0-based channel for the recruitment
ratio feature; also drives table selection.
- ``model_type_ml`` — ``'xgboost'``, ``'logistic_regression'``,
``'random_forest'``.
- ``positive_control`` / ``negative_control`` — well IDs (e.g.
``'c2'`` / ``'c1'``) used as training labels.
- ``annotation_column`` — override controls with a PNG-level
annotation column.
- ``location_column`` — ``'columnID'`` or ``'rowID'``.
- ``heatmap_feature`` — feature plotted on the plate heatmap.
- ``exclude``, ``n_repeats``, ``top_features``, ``test_size``,
``reg_alpha``, ``reg_lambda``, ``learning_rate``,
``n_estimators``, ``n_jobs``.
- ``remove_low_variance_features``,
``remove_highly_correlated_features``, ``prune_features``,
``cross_validation``, ``verbose``.
:returns: The two-element list ``[output, plate_heatmap]``, where
``output`` is the 10-element result list of :func:`ml_analysis`
and ``plate_heatmap`` is the plate-heatmap ``matplotlib``
figure. The CSVs and figures are written to ``results/`` as a
side effect; their paths are not returned.
:raises ValueError: if ``annotation_column`` is set but the
``png_list`` table lacks ``prcfo`` / that column, its object IDs do
not join to the measurements, it contains fewer than two observed
classes, or if ``heatmap_feature`` is not among the trained features.
Example:
.. code-block:: python
from spacr.ml import generate_ml_scores
settings = {
'src': '/data/plate01',
'channel_of_interest': 3,
'positive_control_id': 'c2', 'negative_control_id': 'c1',
'model_type_ml': 'xgboost', 'heatmap_feature': 'recruitment',
}
generate_ml_scores(settings)
See Also:
:func:`ml_analysis` — the underlying fit/evaluate routine.
:func:`perform_regression` — mixed-effects regression on
per-well ML scores.
"""
from .io import _read_and_merge_data, _read_db
from .plot import plot_plates
from .utils import (get_ml_results_paths, calculate_shortest_distance,
save_settings, _measurement_store_for)
from .settings import set_default_analyze_screen
from .predictions import (ML_CLASS_COLUMN, merge_ml_predictions,
migrate_prediction_columns)
settings = set_default_analyze_screen(settings)
save_settings(settings, name='generate_ml_scores', show=True)
_flowview_advance("tables")
srcs = settings['src']
if isinstance(srcs, str):
srcs = [srcs]
df = pd.DataFrame()
measurement_stores = []
for idx, src in enumerate(srcs):
if idx == 0:
src1 = src
sqlite_path = os.path.join(src, 'measurements', 'measurements.db')
db_loc = [_measurement_store_for(sqlite_path, settings) or sqlite_path]
if (str(settings.get('measurement_backend') or 'sqlite').lower() != 'sqlite'
and db_loc[0] in measurement_stores):
continue
measurement_stores.append(db_loc[0])
tables = ['cell', 'nucleus', 'pathogen','cytoplasm']
dft, _ = _read_and_merge_data(db_loc,
tables,
settings['verbose'],
nuclei_limit=settings['nuclei_limit'],
pathogen_limit=settings['pathogen_limit'])
df = pd.concat([df, dft])
_flowview_metric("objects", len(df))
_flowview_metric("databases", len(measurement_stores))
_flowview_metric("tables", len(tables) * len(measurement_stores))
try:
df = calculate_shortest_distance(df, 'pathogen', 'nucleus')
except Exception as e:
print(e)
from .training_basis import resolve_basis
_basis = resolve_basis(settings)
#: The column the annotation path trains against. None on the metadata
#: path, where the caller's own `location_column` is the answer. Declared
#: here so every branch below has it defined.
_label_column = None
if _basis == 'annotation':
if not settings.get('annotation_column'):
raise ValueError(
"dataset_mode='annotation' needs annotation_column set to a "
"column of png_list. Nothing else in these settings says "
"which labels to train on.")
_label_column = settings['annotation_column']
migrate_prediction_columns(db_loc[0])
png_list_df = _read_db(db_loc[0], tables=['png_list'])[0]
if not {'prcfo', settings['annotation_column']}.issubset(png_list_df.columns):
raise ValueError("The 'png_list_df' DataFrame must contain 'prcfo' and 'test' columns.")
annotated_df = png_list_df[['prcfo', settings['annotation_column']]].set_index('prcfo')
measurement_rows = len(df)
annotation_rows = len(annotated_df)
df = annotated_df.merge(df, left_index=True, right_index=True,
validate='many_to_one')
if df.empty:
raise ValueError(
f"annotation_column={settings['annotation_column']!r} joined "
f"to 0 measured objects by 'prcfo' ({annotation_rows} "
f"annotation rows; {measurement_rows} measurement rows), so "
f"there is no training data. Verify that png_list and the "
f"measurement tables come from the same source and use the "
f"same object identities.")
unique_values = df[settings['annotation_column']].dropna().unique()
print(f"Unique values in annotation column: {unique_values}")
if len(unique_values) < 2:
labelled_rows = int(
df[settings['annotation_column']].notna().sum())
if not len(unique_values):
state = (f"has 0 non-empty labels across {len(df)} joined "
f"object rows")
else:
state = (f"has only one observed class across "
f"{labelled_rows} labelled object rows")
raise ValueError(
f"annotation_column={settings['annotation_column']!r} "
f"{state}; binary ML training requires two real annotated "
f"classes. Annotate objects in a second class, or choose the "
f"annotation column that already contains both classes. "
f"Unannotated objects will be scored after training; spaCR "
f"will not assign them a training label.")
if settings['positive_control_id'] is None and settings['negative_control_id'] is None:
settings['positive_control_id'] = str(unique_values[0])
settings['negative_control_id'] = str(unique_values[1])
print(f"Automatically set positive control to {settings['positive_control_id']} and negative control to {settings['negative_control_id']} based on unique values in annotation column.")
_flowview_advance("dataset")
from .utils import feature_selection
recruitment_channel = feature_selection(settings['channel_of_interest'])
if isinstance(recruitment_channel, int):
pathogen_col = f"pathogen_channel_{recruitment_channel}_mean_intensity"
cytoplasm_col = f"cytoplasm_channel_{recruitment_channel}_mean_intensity"
if pathogen_col in df.columns and cytoplasm_col in df.columns:
df['recruitment'] = df[pathogen_col]/df[cytoplasm_col]
from .batch_correction import correction_kwargs
batch_kwargs = correction_kwargs(
settings,
default_control_column=(_label_column
or settings.get('location_column')),
default_control_values=settings.get('negative_control_id'),
)
batch_kwargs['batch_covariate_column'] = settings.get(
'batch_covariate_column')
batch_kwargs['batch_combat_mean_only'] = bool(
settings.get('batch_combat_mean_only', False))
_training_column = _label_column or settings['location_column']
output, figs = ml_analysis(df,
settings['channel_of_interest'],
_training_column,
settings['positive_control_id'],
settings['negative_control_id'],
settings['exclude'],
settings['n_repeats'],
settings['top_features'],
settings['reg_alpha'],
settings['reg_lambda'],
settings['learning_rate'],
settings['n_estimators'],
settings['test_size'],
settings['model_type_ml'],
settings['n_jobs'],
settings['remove_low_variance_features'],
settings['remove_highly_correlated_features'],
settings['prune_features'],
settings['cross_validation'],
settings['verbose'],
split_by=settings.get('cv_group_by', 'well'),
holdout_plate=settings.get('holdout_plate'),
**batch_kwargs)
shap_fig = shap_analysis(output[3], output[4], output[5])
features = output[0].select_dtypes(include=[np.number]).columns.tolist()
train_features_df = pd.DataFrame(output[9], columns=['feature'])
if not settings['heatmap_feature'] in features:
raise ValueError(f"Variable {settings['heatmap_feature']} not found in the dataframe. Please choose one of the following: {features}")
plate_heatmap = plot_plates(df=output[0],
variable=settings['heatmap_feature'],
grouping=settings['grouping'],
min_max=settings['min_max'],
cmap=settings['cmap'],
min_count=settings['min_cells_per_well'],
verbose=settings['verbose'])
data_path, permutation_path, feature_importance_path, model_metricks_path, permutation_fig_path, feature_importance_fig_path, shap_fig_path, plate_heatmap_path, settings_csv, ml_features = get_ml_results_paths(src1, settings['model_type_ml'], settings['channel_of_interest'])
df, permutation_df, feature_importance_df, _, _, _, _, _, metrics_df, _ = output
_flowview_metric("objects", len(output[0]))
_flowview_metric("test_objects", len(output[5]))
_flowview_advance("scores")
df.to_csv(data_path, mode='w', encoding='utf-8')
permutation_df.to_csv(permutation_path, mode='w', encoding='utf-8')
feature_importance_df.to_csv(feature_importance_path, mode='w', encoding='utf-8')
train_features_df.to_csv(ml_features, mode='w', encoding='utf-8')
metrics_df.to_csv(model_metricks_path, mode='w', encoding='utf-8')
from .figure_sink import publish
plate_heatmap_path = publish(plate_heatmap, plate_heatmap_path)
permutation_fig_path = publish(figs[0], permutation_fig_path)
feature_importance_fig_path = publish(
figs[1], feature_importance_fig_path)
shap_fig_path = write_plot(shap_fig, shap_fig_path, "SHAP summary")
settings['csv_path'] = data_path
settings['db_path'] = measurement_stores[0]
settings['table_name'] = 'png_list'
settings['update_column'] = ML_CLASS_COLUMN
settings['match_column'] = 'prcfo'
matched_objects = 0
unmatched_objects = 0
for store in measurement_stores:
report = merge_ml_predictions(
df,
store,
table=settings['table_name'],
)
if report is not None:
matched_objects += report.matched_rows
unmatched_objects += report.unmatched_db_rows
_flowview_metric("objects", len(df))
_flowview_metric("matched_objects", matched_objects)
_flowview_metric("unmatched_objects", unmatched_objects)
_flowview_metric("databases", len(measurement_stores))
return [output, plate_heatmap]
def _resolve_controls(df, location_column, negative_control,
positive_control, matches):
"""The control values to match, and whether they had to be derived.
:param matches: ``(series, control) -> boolean mask``. Passed in rather
than imported because the matcher is defined inside `ml_analysis`;
taking it as an argument keeps this function module-level and
testable on its own.
:returns: ``(negative, positive, derived)``. ``derived`` is True when the
named controls matched nothing and the column's own two classes were
used instead.
THE CASE THIS EXISTS FOR: annotation mode points `location_column` at the
annotation column, whose values are class labels, while the control
settings still hold plate column names from the metadata path. Neither
matches, and the user is told to "set positive_control and
negative_control to values that appear there" -- for a column that
already says, unambiguously, what its two classes are.
Nothing is derived when the named controls DO match: an explicit choice
is always honoured, including a deliberate two-of-five subset.
"""
if location_column not in df.columns:
return negative_control, positive_control, False
column = df[location_column]
if isinstance(column, pd.DataFrame):
return negative_control, positive_control, False
any_found = (matches(column, negative_control).any()
or matches(column, positive_control).any())
if any_found:
return negative_control, positive_control, False
present = sorted(v for v in column.dropna().unique())
if len(present) != 2:
return negative_control, positive_control, False
low, high = present
print(f"{location_column!r} holds exactly two classes, {low!r} and "
f"{high!r}, and neither {negative_control!r} nor "
f"{positive_control!r} appears in it. Training on the column's own "
f"classes: negative={low!r}, positive={high!r}.")
return low, high, True
@single_threaded_openmp('classical ML training')
[docs]
def ml_analysis(
df,
channel_of_interest=3,
location_column='columnID',
positive_control='c2',
negative_control='c1',
exclude=None,
n_repeats=10,
top_features=30,
reg_alpha=0.1,
reg_lambda=1.0,
learning_rate=0.00001,
n_estimators=1000,
test_size=0.2,
model_type='xgboost',
n_jobs=-1,
remove_low_variance_features=True,
remove_highly_correlated_features=True,
prune_features=False,
cross_validation=False,
verbose=False,
*,
split_by='well',
holdout_plate=None,
batch_correction='none',
batch_column='plateID',
batch_control_column=None,
batch_control_values=None,
batch_covariate_column=None,
batch_combat_mean_only=False,
batch_min_samples=3,
batch_missing_control='error',
):
"""Train a per-object classifier on positive/negative control wells and score every row of the input DataFrame.
Called directly for one-off ML work, and internally by
:func:`generate_ml_scores`. Filters features by channel, drops
low-variance and highly correlated columns, splits (or CVs) train /
test, fits the requested model, computes permutation and native
feature importances, tunes an optimal decision threshold and writes
predictions + probabilities back onto the returned DataFrame.
:param df: Per-object feature DataFrame as produced by merging the
cell/nucleus/pathogen/cytoplasm tables of a
:func:`spacr.measure.measure_crop` database.
:param channel_of_interest: Channel index used to select features.
:param location_column: Column identifying wells / plate columns.
Default ``'columnID'``.
:param positive_control: Value(s) in ``location_column`` treated as
the positive class. Default ``'c2'``.
:param negative_control: Value(s) treated as the negative class.
Default ``'c1'``.
:param exclude: Columns to remove from feature space.
:param n_repeats: Repeats for permutation importance. Default ``10``.
:param top_features: Feature cap when ``prune_features=True``.
:param reg_alpha: XGBoost L1 penalty.
:param reg_lambda: XGBoost L2 penalty.
:param learning_rate: XGBoost learning rate.
:param n_estimators: Tree count for tree-based models.
:param test_size: Test-split fraction. Default ``0.2``.
:param model_type: ``'random_forest'``, ``'logistic_regression'``,
``'gradient_boosting'`` or ``'xgboost'``.
:param n_jobs: Parallel job count where applicable. Default ``-1``.
:param remove_low_variance_features: Drop low-variance features.
:param remove_highly_correlated_features: Drop highly correlated features.
:param prune_features: If True, apply ``SelectKBest`` before training.
:param cross_validation: If True, run 5-fold stratified CV.
:param verbose: Log progress details.
:param split_by: Independent acquisition unit for train/test splitting:
``'cell'``, ``'field'``, ``'well'`` (default), or ``'plate'``.
Legacy ``'none'`` is an alias for ``'cell'``.
:param batch_correction: plate correction method from
:mod:`spacr.batch_correction`.
:param batch_column: metadata column identifying plates/batches.
:param batch_control_column: metadata column holding reference-control
labels for ``control_center``.
:param batch_control_values: negative/reference control value(s).
:param batch_min_samples: minimum rows or controls per plate.
:param batch_covariate_column: Metadata column containing a biological
covariate that ComBat must preserve, such as treatment, cell line, or
time point. Required when ``batch_correction="combat"``; its
coefficients remain in the corrected data while estimated batch
effects are removed.
:param batch_combat_mean_only: If ``True``, ComBat adjusts batch means
without scaling batch variances. This can be appropriate when batches
differ primarily by location or contain too few observations for
stable variance estimates. Default ``False`` adjusts both means and
variances.
:param batch_missing_control: ``error`` or ``skip`` for missing controls.
:returns: Tuple ``(output, figs)`` where ``output`` is a positional
tuple of ``(scored_df, permutation_df, feature_importance_df,
model, X_train, X_test, y_train, y_test, metrics_df,
train_features)`` and ``figs`` is
``(permutation_fig, feature_importance_fig)``.
:raises ValueError: on unsupported ``model_type`` or when positive /
negative control rows cannot be located in ``location_column``.
Example:
.. code-block:: python
from spacr.ml import ml_analysis
output, figs = ml_analysis(
df, channel_of_interest=3,
positive_control='c2', negative_control='c1',
model_type='xgboost',
)
scored_df = output[0]
See Also:
:func:`generate_ml_scores` — wraps this call with DB I/O.
"""
from .resource_log import _guard_workers, _table_nbytes
n_jobs = _guard_workers('ml_analyze', n_jobs, _table_nbytes(df))
_flowview_advance("dataset")
def _match_control_values(series, control):
"""
Return a boolean mask selecting rows in `series` that match `control`.
Matching is attempted in this order:
1. exact value match
2. numeric coercion match
3. stripped string match
`control` can be a scalar or a list/tuple/set of values.
"""
if isinstance(control, (list, tuple, set, np.ndarray, pd.Series)):
controls = list(control)
else:
controls = [control]
mask = pd.Series(False, index=series.index)
for c in controls:
current_mask = pd.Series(False, index=series.index)
try:
current_mask |= (series == c)
except Exception:
pass
try:
s_num = pd.to_numeric(series, errors='coerce')
c_num = pd.to_numeric(pd.Series([c]), errors='coerce').iloc[0]
if pd.notna(c_num):
current_mask |= (s_num == c_num)
except Exception:
pass
try:
s_str = series.astype(str).str.strip()
c_str = str(c).strip()
current_mask |= (s_str == c_str)
except Exception:
pass
mask |= current_mask
return mask
from .utils import filter_dataframe_features
from .plot import plot_permutation, plot_feature_importance
random_state = _run_random_state(42)
if 'cells_per_well' in df.columns:
df = df.drop(columns=['cells_per_well'])
correction_metadata = df.copy()
if location_column not in df.columns:
available = ", ".join(repr(c) for c in list(df.columns)[:12])
if len(df.columns) > 12:
available += f", ... ({len(df.columns)} columns)"
raise ValueError(
f"location_column={location_column!r} is not a column of the "
f"measurement table, so there is nothing to group the controls "
f"by.\n The table has: {available}"
f"\n If you have run this module in annotation mode, that is "
f"the likely cause: versions before 1.5.0.5 wrote "
f"annotation_column into location_column and never put it back, "
f"so a later metadata run looked for an annotation column in the "
f"measurement table. Set location_column back to your well "
f"column ('columnID' or 'rowID').")
if df.empty:
raise ValueError(
"the measurement table contains 0 object rows, so there is "
"nothing to train on. Check that the selected source contains "
"measured objects before running the analysis.")
location_values = df[location_column]
if isinstance(location_values, pd.Series):
non_empty_values = location_values.dropna().astype(str).str.strip()
if not non_empty_values.ne("").any():
raise ValueError(
f"location_column={location_column!r} has 0 non-empty values "
f"across {len(df)} object rows. Populate it with two real "
f"class labels before running the analysis.")
df_metadata = df[[location_column]].copy()
df, features = filter_dataframe_features(df, channel_of_interest, exclude, remove_low_variance_features, remove_highly_correlated_features, verbose)
print('After filtration:', len(df))
if str(batch_correction or 'none').strip().lower() not in {
'none', 'off', 'false',
}:
from .batch_correction import correct_from_metadata
corrected, correction_report = correct_from_metadata(
df[features],
correction_metadata.loc[df.index],
batch_correction=batch_correction,
batch_column=batch_column,
batch_control_column=batch_control_column,
batch_control_values=batch_control_values,
batch_covariate_column=batch_covariate_column,
batch_combat_mean_only=batch_combat_mean_only,
batch_min_samples=batch_min_samples,
batch_missing_control=batch_missing_control,
)
df.loc[:, features] = corrected
print(
f"Batch correction {correction_report.method}: "
f"{correction_report.centroid_spread_before} -> "
f"{correction_report.centroid_spread_after} centroid spread.")
for note in correction_report.warnings:
print(f"Warning: batch correction: {note}")
if verbose:
print(f'Found {len(features)} numerical features in the dataframe')
print(f'Features used in training: {features}')
print(f'Features: {features}')
df = pd.concat([df, df_metadata[location_column]], axis=1)
df['prcfo'] = df.index.astype(str)
negative_control, positive_control, _derived_classes = _resolve_controls(
df, location_column, negative_control, positive_control,
_match_control_values)
df1 = df[_match_control_values(df[location_column], negative_control)].copy()
if verbose:
print(f'Negative control: {negative_control}, samples: {len(df1)}')
df2 = df[_match_control_values(df[location_column], positive_control)].copy()
if verbose:
print(f'Positive control: {positive_control}, samples: {len(df2)}')
df1['target'] = 0
df2['target'] = 1
combined_df = pd.concat([df1, df2])
combined_df = combined_df.drop(columns=[location_column])
if verbose:
print(f'Found {len(df1)} samples for {negative_control} and {len(df2)} samples for {positive_control}. Total: {len(combined_df)}')
untrained = []
try:
column = df[location_column] if location_column in df.columns else None
if isinstance(column, pd.Series):
trained_on = set(df1[location_column].unique()) | set(
df2[location_column].unique())
present = set(column.dropna().unique())
untrained = sorted(str(value) for value in present - trained_on)
except Exception: # noqa: BLE001
LOG.debug("could not list the classes outside the training set",
exc_info=True)
if untrained:
print(f"{len(untrained)} class(es) of {location_column!r} are not in "
f"the training set and are SCORED by a model that never saw "
f"them: {untrained[:10]}"
f"{'...' if len(untrained) > 10 else ''}. This fit is binary: "
f"one arm is negative_control_id={negative_control!r} and the "
f"other positive_control_id={positive_control!r}. Both take a "
f"list, so name several values to pool them into one arm.")
if df1.empty or df2.empty:
column = df[location_column]
if isinstance(column, pd.DataFrame):
raise ValueError(
f"the measurement table has {column.shape[1]} columns named "
f"{location_column!r}, so the controls cannot be matched "
f"against it. Drop or rename the duplicate before running "
f"the analysis.")
present = column.astype(str).str.strip().unique().tolist()
shown = ", ".join(repr(v) for v in sorted(present)[:15])
if len(present) > 15:
shown += f", ... ({len(present)} distinct values)"
missing = []
if df1.empty:
missing.append(f"negative_control_id={negative_control!r}")
if df2.empty:
missing.append(f"positive_control_id={positive_control!r}")
raise ValueError(
f"no rows matched {' and '.join(missing)} in column "
f"{location_column!r}, so there is nothing to train on.\n"
f" {location_column!r} contains: {shown}\n"
f" Set positive_control_id and negative_control_id to values "
f"that appear there, or set location_column to the column that "
f"holds your controls.")
X = combined_df[features]
y = combined_df['target']
if prune_features:
before_pruning = len(X.columns)
selector = SelectKBest(score_func=f_classif, k=top_features)
X_selected = selector.fit_transform(X, y)
selected_features = X.columns[selector.get_support()]
X = pd.DataFrame(X_selected, columns=selected_features, index=X.index)
features = selected_features.tolist()
after_pruning = len(X.columns)
print(f"Removed {before_pruning - after_pruning} features using SelectKBest")
_flowview_metric("objects", len(df))
_flowview_metric("training_objects", len(combined_df))
_flowview_metric("features", len(features))
_flowview_advance("split")
from .classifier_evaluation import grouped_split, split_group_values
split_frame = combined_df[['prcfo']].reset_index(drop=True)
split_level, split_groups = split_group_values(
group_by=split_by, frame=split_frame, table='ML control measurements')
held = holdout_plate
if held is not None and not isinstance(held, (list, tuple, set)):
held = [held]
if held:
_plate_level, plate_groups = split_group_values(
group_by='plate', frame=split_frame,
table='ML control measurements')
train_index, test_index, split_report = grouped_split(
plate_groups, y.to_numpy(), test_size, seed=random_state,
group_by='plate', hold_out_groups=held)
else:
train_index, test_index, split_report = grouped_split(
split_groups, y.to_numpy(), test_size, seed=random_state,
group_by=split_level)
X_train, X_test = X.iloc[train_index], X.iloc[test_index]
y_train, y_test = y.iloc[train_index], y.iloc[test_index]
print(split_report.summary())
combined_df['data_usage'] = 'train'
combined_df.loc[X_test.index, 'data_usage'] = 'test'
df['data_usage'] = 'not_used'
df.loc[combined_df.index, 'data_usage'] = combined_df['data_usage']
df['data_usage_group_by'] = split_report.group_by
df['split_requested_fraction'] = split_report.requested_fraction
df['split_cell_fraction'] = split_report.cell_fraction
df['split_group_fraction'] = split_report.group_fraction
_flowview_metric("objects", len(X))
_flowview_metric("train_objects", len(X_train))
_flowview_metric("test_objects", len(X_test))
_flowview_advance("model")
if model_type == 'random_forest':
model = RandomForestClassifier(n_estimators=n_estimators, random_state=random_state, n_jobs=n_jobs)
elif model_type == 'extra_trees':
from sklearn.ensemble import ExtraTreesClassifier
model = ExtraTreesClassifier(n_estimators=n_estimators, random_state=random_state, n_jobs=n_jobs)
elif model_type == 'logistic_regression':
model = LogisticRegression(max_iter=1000, random_state=random_state)
elif model_type == 'gradient_boosting':
model = HistGradientBoostingClassifier(max_iter=n_estimators, random_state=random_state)
elif model_type == 'xgboost':
model = XGBClassifier(
reg_alpha=reg_alpha,
reg_lambda=reg_lambda,
learning_rate=learning_rate,
n_estimators=n_estimators,
random_state=random_state,
nthread=n_jobs,
eval_metric='logloss',
)
elif model_type == 'lightgbm':
try:
from lightgbm import LGBMClassifier
except ImportError:
raise ImportError("model_type='lightgbm' requires the 'lightgbm' package. Install it with: pip install lightgbm")
model = LGBMClassifier(n_estimators=n_estimators, learning_rate=learning_rate, reg_alpha=reg_alpha, reg_lambda=reg_lambda, random_state=random_state, n_jobs=n_jobs)
elif model_type == 'catboost':
try:
from catboost import CatBoostClassifier
except ImportError:
raise ImportError("model_type='catboost' requires the 'catboost' package. Install it with: pip install catboost")
model = CatBoostClassifier(iterations=n_estimators, learning_rate=learning_rate, l2_leaf_reg=reg_lambda, random_state=random_state, thread_count=n_jobs, verbose=False)
elif model_type == 'svm':
from sklearn.calibration import CalibratedClassifierCV
from sklearn.svm import SVC
model = CalibratedClassifierCV(
estimator=SVC(random_state=random_state),
method='sigmoid',
cv=3,
n_jobs=n_jobs,
ensemble=False,
)
elif model_type == 'mlp':
from sklearn.neural_network import MLPClassifier
model = MLPClassifier(max_iter=max(200, n_estimators), random_state=random_state)
else:
raise ValueError(f"Unsupported model_type: {model_type}")
model.spacr_split_report_ = split_report.to_dict()
_flowview_metric("features", len(X.columns))
_flowview_advance("training")
if cross_validation:
from .io import make_cv_folds
distinct_groups = len(np.unique(split_groups))
n_folds = min(5, distinct_groups)
folds = make_cv_folds(
y.to_numpy(), n_folds, groups=split_groups,
seed=random_state)
expected_classes = set(np.unique(y))
fold_metrics = []
for fold_idx, (train_index, test_index) in enumerate(folds, start=1):
if (set(np.unique(y.iloc[train_index])) != expected_classes or
set(np.unique(y.iloc[test_index])) != expected_classes):
raise ValueError(
f"{split_level}-grouped CV fold {fold_idx} cannot put "
"every class in both train and test. Add independent "
f"class-bearing {split_level}s or choose a finer split.")
X_train, X_test = X.iloc[train_index], X.iloc[test_index]
y_train, y_test = y.iloc[train_index], y.iloc[test_index]
model.fit(X_train, y_train)
predictions_test = model.predict(X_test)
combined_df.loc[X_test.index, 'predictions'] = predictions_test
prediction_probabilities_test = model.predict_proba(X_test)
optimal_threshold = find_optimal_threshold(y_test, prediction_probabilities_test[:, 1])
if verbose:
print(f'Fold {fold_idx} - Optimal threshold: {optimal_threshold}')
df.loc[X_test.index, 'predictions'] = predictions_test
for i in range(prediction_probabilities_test.shape[1]):
df.loc[X_test.index, f'prediction_probability_class_{i}'] = prediction_probabilities_test[:, i]
fold_report = classification_report(
y_test, predictions_test, output_dict=True, zero_division=0)
fold_metrics.append(pd.DataFrame(fold_report).transpose())
if verbose:
print(f"Fold {fold_idx} Classification Report:")
print(classification_report(
y_test, predictions_test, zero_division=0))
metrics_df = pd.concat(fold_metrics).groupby(level=0).mean()
model.fit(X, y)
all_predictions = model.predict(df[features])
df['predictions'] = all_predictions
prediction_probabilities = model.predict_proba(df[features])
for i in range(prediction_probabilities.shape[1]):
df[f'prediction_probability_class_{i}'] = prediction_probabilities[:, i]
else:
model.fit(X_train, y_train)
predictions_test = model.predict(X_test)
combined_df.loc[X_test.index, 'predictions'] = predictions_test
prediction_probabilities_test = model.predict_proba(X_test)
optimal_threshold = find_optimal_threshold(y_test, prediction_probabilities_test[:, 1])
if verbose:
print(f'Optimal threshold: {optimal_threshold}')
X_all = df[features]
all_predictions = model.predict(X_all)
df['predictions'] = all_predictions
prediction_probabilities = model.predict_proba(X_all)
for i in range(prediction_probabilities.shape[1]):
df[f'prediction_probability_class_{i}'] = prediction_probabilities[:, i]
if verbose:
print("\nClassification Report:")
print(classification_report(
y_test, predictions_test, zero_division=0))
report_dict = classification_report(
y_test, predictions_test, output_dict=True, zero_division=0)
metrics_df = pd.DataFrame(report_dict).transpose()
_flowview_metric("objects", len(X))
_flowview_metric("features", len(features))
_flowview_advance("evaluation")
metrics_df['split_group_by'] = split_report.group_by
metrics_df['split_requested_fraction'] = split_report.requested_fraction
metrics_df['split_group_fraction'] = split_report.group_fraction
metrics_df['split_cell_fraction'] = split_report.cell_fraction
perm_importance = permutation_importance(model, X_train, y_train, n_repeats=n_repeats, random_state=random_state, n_jobs=guarded_n_jobs(n_jobs, 'permutation importance'))
permutation_df = pd.DataFrame({
'feature': [features[i] for i in perm_importance.importances_mean.argsort()],
'importance_mean': perm_importance.importances_mean[perm_importance.importances_mean.argsort()],
'importance_std': perm_importance.importances_std[perm_importance.importances_mean.argsort()]
}).tail(top_features)
permutation_fig = plot_permutation(permutation_df)
if verbose:
permutation_fig.show()
if hasattr(model, 'feature_importances_'):
feature_importances = model.feature_importances_
feature_importance_df = pd.DataFrame({
'feature': features,
'importance': feature_importances
}).sort_values(by='importance', ascending=False).head(top_features)
feature_importance_fig = plot_feature_importance(feature_importance_df)
if verbose:
feature_importance_fig.show()
else:
feature_importance_df = permutation_df.rename(
columns={"importance_mean": "importance"}
)[["feature", "importance"]].sort_values(
by="importance", ascending=False).head(top_features)
feature_importance_fig = plot_feature_importance(
feature_importance_df,
title=f"Top {len(feature_importance_df)} features "
f"(permutation importance)")
if verbose:
feature_importance_fig.show()
df = _calculate_similarity(df, features, location_column, positive_control, negative_control)
df['prcfo'] = df.index.astype(str)
df = _assign_prcfo_parts(df, object_column='object')
df['prc'] = _compose_prc_column(df)
return [df, permutation_df, feature_importance_df, model, X_train, X_test, y_train, y_test, metrics_df, features], [permutation_fig, feature_importance_fig]
#: How many background rows a model-agnostic explainer is given. The
#: permutation explainer is O(background x features) per explained row, so
#: the whole training set turns a 0.3 s panel into minutes. Summarising the
#: background is what the SHAP authors recommend for exactly this.
SHAP_BACKGROUND = 100
def _shap_values(model, X_train, X_test):
"""(values, note). SHAP contributions for every model the panel offers.
THE FAILURE IS NOT ALWAYS AT CONSTRUCTION. `shap.Explainer(model, X)`
accepts an xgboost booster happily and raises "Categorical split is not
yet supported" only when it is CALLED -- so a fallback chosen at
construction time never ran, and the panel's DEFAULT model produced no
SHAP at all. Each candidate is therefore tried all the way through.
"""
import shap
attempts = list(_shap_explainers(model, X_train))
trouble = ""
for explainer, note in attempts:
try:
return explainer(X_test), note
except Exception as error: # noqa: BLE001
trouble = f"{type(error).__name__}: {error}"
LOG.debug("a SHAP explainer would not run", exc_info=True)
raise RuntimeError(
f"No SHAP explainer could explain {type(model).__name__}. The last "
f"failure was: {trouble}")
def _shap_explainers(model, X_train):
"""Every explainer worth trying for this estimator, best first.
THREE OF THE NINE MODELS THE PANEL OFFERS COULD NOT BE EXPLAINED AT
ALL, including the default:
* `xgboost` raised "Categorical split is not yet supported. You can
still use TreeExplainer with feature_perturbation=tree_path_dependent"
-- an error carrying its own fix, which nothing acted on.
* `svm` and `mlp` raised "The passed model is not callable and cannot be
analyzed directly with the given masker". A support vector machine and
a neural net are not trees and not linear; the model-agnostic
explainer takes a FUNCTION, not an estimator.
The note is returned rather than printed here so the caller decides
where it goes, and it is said out loud because the three explainers do
not compute the same quantity: `tree_path_dependent` conditions on the
tree's own splits rather than on an independent background, and the
permutation explainer estimates rather than solves.
"""
import shap
try:
automatic_note = ""
if type(model).__module__.split(".", 1)[0] == "xgboost":
automatic_note = (
"SHAP: XGBoost was accepted by the automatic tree "
"explainer, so the panel shows tree contributions rather "
"than a model-agnostic estimate."
)
yield shap.Explainer(model, X_train), automatic_note
except Exception: # noqa: BLE001
LOG.debug("the default SHAP explainer would not build",
exc_info=True)
try:
yield (shap.TreeExplainer(
model, feature_perturbation="tree_path_dependent"),
"SHAP: this model has categorical splits, so it is explained "
"with feature_perturbation='tree_path_dependent' -- which "
"conditions on the tree's own splits rather than on an "
"independent background.")
except Exception: # noqa: BLE001
LOG.debug("this model is not a tree", exc_info=True)
predict = (getattr(model, "predict_proba", None)
or getattr(model, "decision_function", None)
or getattr(model, "predict", None))
if predict is None:
return
background = X_train
if hasattr(X_train, "shape") and X_train.shape[0] > SHAP_BACKGROUND:
background = shap.utils.sample(X_train, SHAP_BACKGROUND,
random_state=0)
try:
yield (shap.Explainer(predict, background),
f"SHAP: {type(model).__name__} is neither a tree nor a "
f"linear model, so it is explained through its predictions "
f"over {len(background)} background row(s). That is an "
f"ESTIMATE of each contribution rather than an exact "
f"decomposition.")
except Exception: # noqa: BLE001
LOG.debug("the model-agnostic SHAP explainer would not build",
exc_info=True)
[docs]
def shap_analysis(model, X_train, X_test):
"""Build a SHAP summary beeswarm for ``X_test``.
The beeswarm is rendered with pyqtgraph so it can be embedded in the same
scene-based figure workflow as other model-explanation plots.
The function returns a live
:class:`~spacr.qt.widgets.fast_plots.FastPlot`; it neither writes a file
nor returns a matplotlib figure. Pass the result to :func:`write_plot` to
export it in the configured figure format.
:param model: Fitted estimator compatible with ``shap.Explainer``.
:param X_train: Training features used to seed the explainer.
:param X_test: Test features to explain.
:returns: A ``FastPlot`` holding the beeswarm, or ``None`` when Qt is
unavailable or the attribution matrix cannot be plotted.
"""
import shap
shap_values, note = _shap_values(model, X_train, X_test)
if note:
print(note)
if len(shap_values.shape) == 3:
output_index = 1 if shap_values.shape[-1] > 1 else 0
shap_values = shap_values[..., output_index]
from .figures.headless import application
application_object, refusal = application()
if application_object is None:
print(refusal)
return None
from .qt.widgets.fast_plots import FastPlot
matrix = np.asarray(shap_values.values, dtype=float)
if matrix.ndim != 2 or not matrix.size:
return None
columns = list(X_test.columns)[:matrix.shape[1]]
order = np.argsort(np.nanmean(np.abs(matrix), axis=0))[::-1]
names = [str(columns[int(i)]) for i in order]
plot = FastPlot(title="SHAP summary", x_label="SHAP value", y_label="")
plot.resize(1200, max(420, 34 * len(names) + 140))
if not plot.add_beeswarm(names, matrix[:, order],
X_test[names].to_numpy(dtype=float)):
plot.deleteLater()
return None
application_object.processEvents()
return plot
[docs]
def write_plot(plot, path, title=""):
"""Write a pyqtgraph plot out and announce it, like ``publish`` does.
The counterpart of :func:`spacr.figure_sink.publish` for a scene rather
than a matplotlib figure: the format follows the user's preference, the
file NAME follows the format, and the written file reaches the gallery,
because saved and visible are the same event.
``None`` writes nothing, announces nothing and returns None -- a plot
that could not be built must not take the run down after the model has
been fitted and every object scored.
:param plot: a ``FastPlot``, or None.
:param path: where to write it; the extension may be rewritten.
:param title: the name the gallery tile carries.
:returns: the path written, or None.
"""
if plot is None:
return None
from .figure_sink import publish_file
from .plot import figure_output_preferences
chosen = str(figure_output_preferences()[0]).lower().lstrip('.')
stem, _ = os.path.splitext(str(path))
target = f"{stem}.{chosen}"
parent = os.path.dirname(os.path.abspath(target))
os.makedirs(parent, exist_ok=True)
try:
written = plot.export(target)
finally:
plot.deleteLater()
if written:
publish_file(written, title=title or None)
return written
[docs]
def find_optimal_threshold(y_true, y_pred_proba):
"""Return the probability threshold maximising F1 on the precision-recall curve.
:param y_true: Ground-truth binary labels.
:param y_pred_proba: Predicted probabilities for the positive class.
:returns: Optimal probability threshold.
"""
precision, recall, thresholds = precision_recall_curve(y_true, y_pred_proba)
denominator = precision + recall
with np.errstate(divide='ignore', invalid='ignore'):
f1_scores = np.where(denominator > 0,
2 * (precision * recall) / denominator,
0.0)
optimal_idx = np.argmax(f1_scores)
optimal_threshold = thresholds[optimal_idx]
return optimal_threshold
def _calculate_similarity(df, features, col_to_compare, val1, val2):
"""
Calculate similarity scores of each well to the positive and negative controls using various metrics.
Args:
df (pandas.DataFrame): DataFrame containing the data.
features (list): List of feature columns to use for similarity calculation.
col_to_compare (str): Column name to use for comparing groups.
val1, val2 (str): Values in col_to_compare to create subsets for comparison.
Returns:
pandas.DataFrame: DataFrame with similarity scores.
"""
if isinstance(val1, str):
pos_control = df[df[col_to_compare] == val1][features].mean()
elif isinstance(val1, list):
pos_control = df[df[col_to_compare].isin(val1)][features].mean()
if isinstance(val2, str):
neg_control = df[df[col_to_compare] == val2][features].mean()
elif isinstance(val2, list):
neg_control = df[df[col_to_compare].isin(val2)][features].mean()
scaler = StandardScaler()
scaled_features = scaler.fit_transform(df[features])
cov_matrix = np.cov(scaled_features, rowvar=False)
inv_cov_matrix = None
try:
inv_cov_matrix = np.linalg.inv(cov_matrix)
except np.linalg.LinAlgError:
epsilon = 1e-5
inv_cov_matrix = np.linalg.inv(cov_matrix + np.eye(cov_matrix.shape[0]) * epsilon)
def safe_similarity(func, row, control, *args, **kwargs):
"""Call ``func(row, control, ...)`` and swallow errors (return ``NaN``)."""
try:
return func(row, control, *args, **kwargs)
except Exception:
return np.nan
try:
df['similarity_to_pos_euclidean'] = df[features].apply(lambda row: safe_similarity(euclidean, row, pos_control), axis=1)
df['similarity_to_neg_euclidean'] = df[features].apply(lambda row: safe_similarity(euclidean, row, neg_control), axis=1)
df['similarity_to_pos_cosine'] = df[features].apply(lambda row: safe_similarity(cosine, row, pos_control), axis=1)
df['similarity_to_neg_cosine'] = df[features].apply(lambda row: safe_similarity(cosine, row, neg_control), axis=1)
df['similarity_to_pos_mahalanobis'] = df[features].apply(lambda row: safe_similarity(mahalanobis, row, pos_control, inv_cov_matrix), axis=1)
df['similarity_to_neg_mahalanobis'] = df[features].apply(lambda row: safe_similarity(mahalanobis, row, neg_control, inv_cov_matrix), axis=1)
df['similarity_to_pos_manhattan'] = df[features].apply(lambda row: safe_similarity(cityblock, row, pos_control), axis=1)
df['similarity_to_neg_manhattan'] = df[features].apply(lambda row: safe_similarity(cityblock, row, neg_control), axis=1)
df['similarity_to_pos_minkowski'] = df[features].apply(lambda row: safe_similarity(minkowski, row, pos_control, p=3), axis=1)
df['similarity_to_neg_minkowski'] = df[features].apply(lambda row: safe_similarity(minkowski, row, neg_control, p=3), axis=1)
df['similarity_to_pos_chebyshev'] = df[features].apply(lambda row: safe_similarity(chebyshev, row, pos_control), axis=1)
df['similarity_to_neg_chebyshev'] = df[features].apply(lambda row: safe_similarity(chebyshev, row, neg_control), axis=1)
df['similarity_to_pos_braycurtis'] = df[features].apply(lambda row: safe_similarity(braycurtis, row, pos_control), axis=1)
df['similarity_to_neg_braycurtis'] = df[features].apply(lambda row: safe_similarity(braycurtis, row, neg_control), axis=1)
except Exception as e:
print(f"Error calculating similarity scores: {e}")
return df
def _announce_the_bundle(folder, title):
"""Put ONE tile in the gallery for a bundle's figure.
ONE PICTURE, ONE TILE. A bundle holds the same figure twice, as a PDF
and as a PNG, because a folder somebody opens should carry both -- but
announcing both puts two tiles in the gallery for one picture, and a
reader clicking each of them to find out they are the same is exactly
the confusion the gallery exists to remove. The one announced is the
one in the format the user chose.
:param folder: the bundle directory.
:param title: the name the tile carries.
:returns: the path announced, or None when the folder holds no figure.
"""
from .figure_sink import publish_file
from .plot import figure_output_preferences
if not folder or not os.path.isdir(folder):
return None
wanted = str(figure_output_preferences()[0]).lower().lstrip('.')
written = sorted(os.listdir(folder))
chosen = next((f for f in written if f.lower().endswith(f".{wanted}")),
None)
if chosen is None:
chosen = next((f for f in written if f.lower().endswith(".pdf")), None)
if chosen is None:
return None
return publish_file(os.path.join(folder, chosen), title)
_EPHEMERAL_FIGURES = None
def _figure_folder(src, save):
"""Where a drawn figure goes: the run folder, or a temporary one.
`save` GATES THE RUN FOLDER, NOT THE PICTURE. Before these charts moved
to pyqtgraph they were `plt.show()`\\ n and never written, so a `save=False`
run still SAW them -- and writing them into the user's results folder
now would be a behaviour change nobody asked for. A temporary directory
is what an ephemeral figure has always been; the gallery gets its tile
either way, because saved and visible are one event.
:param src: plate folder.
:param save: whether this run is writing its results.
:returns: a directory that exists.
"""
global _EPHEMERAL_FIGURES
if save:
folder = os.path.join(str(src), 'results')
os.makedirs(folder, exist_ok=True)
return folder
if _EPHEMERAL_FIGURES is None:
import tempfile
_EPHEMERAL_FIGURES = tempfile.mkdtemp(prefix="spacr-figures-")
return _EPHEMERAL_FIGURES
def _draw_response_panel_in_pyqtgraph(values, transform, column, src):
"""Draw the response distribution before and after, and publish it.
:param values: the untransformed response.
:param transform: the transformation named in the settings.
:param column: the response's own column name.
:param src: plate folder, or None to draw without writing a file.
:returns: the path written, or None.
"""
from .figures.headless import application
application_object, refusal = application()
if application_object is None:
print(refusal)
return None
from .response_distribution import fast_panel
plot = fast_panel(values, transform, dependent_variable=column)
if plot is None:
print("the response distribution panel was not drawn: the "
"response holds no finite values")
return None
plot.resize(1100, 660)
application_object.processEvents()
if not src:
plot.deleteLater()
return None
return write_plot(
plot, os.path.join(_figure_folder(src, True),
'response_distribution.pdf'),
"Response distribution")
def _draw_shap_summary_in_pyqtgraph(shap_values, sample, src, name, top,
save=True):
"""Draw a SHAP beeswarm in pyqtgraph and write its bundle.
The features are ranked by MEAN ABSOLUTE contribution, which is the
order `shap.summary_plot` uses and the only one that answers "which of
these matters": a feature that pushes hard in both directions has a mean
near zero and belongs at the top, not the bottom.
:param shap_values: a shap Explanation, or anything with ``.values``.
:param sample: the frame the values were computed over.
:param src: plate folder; the bundle goes under ``<src>/results``.
:param name: bundle name.
:param top: how many features to show.
:returns: the folder written, or None when there is no Qt.
"""
from .figures.headless import application
from .figure_sink import publish_file
application_object, refusal = application()
if application_object is None:
print(refusal)
return None
from .qt.widgets.fast_plots import FastPlot
from .figures.bundle import save as write_bundle
matrix = np.asarray(getattr(shap_values, 'values', shap_values),
dtype=float)
if matrix.ndim != 2 or not matrix.size:
return None
columns = list(sample.columns)[:matrix.shape[1]]
strength = np.nanmean(np.abs(matrix), axis=0)
order = np.argsort(strength)[::-1][:int(top)]
names = [str(columns[int(i)]) for i in order]
picked = matrix[:, order]
values = sample[names].to_numpy(dtype=float)
title = f"SHAP summary - top {len(names)} features"
plot = FastPlot(title=title, x_label="SHAP value", y_label="")
try:
plot.resize(1200, max(420, 34 * len(names) + 140))
if not plot.add_beeswarm(names, picked, values):
return None
application_object.processEvents()
folder = write_bundle(_figure_folder(src, save), name,
render=plot.export,
data=plot.beeswarm_frame(), groups=None,
unit="observation",
settings={"top_features": int(top),
"figure": name})
finally:
plot.deleteLater()
_announce_the_bundle(folder, title)
return folder
def _draw_the_cell_count_sweep(summary, mark, path):
"""Draw the sample-size sweep in pyqtgraph and write it out.
PUBLISHED, NOT SHOWN. `plt.show()` here blocked forever anywhere there
was no GUI event loop to hand it to: with the Qt backend it calls
`start_main_loop`, and a script, a notebook or `spacr-run regression`
then sat in `qt_compat._exec` until it was killed. Saved and visible are
ONE event, through the figure sink, and a figure reaching the gallery
must not depend on somebody calling `show`.
:param summary: frame with ``sample_size``, ``smoothed_mean_abs_diff``
and ``std_abs_diff``.
:param mark: the sample size the threshold line is drawn at.
:param path: destination; the extension follows the format preference.
:returns: the path written, or None when there is no Qt to draw under.
"""
from .figures.headless import application
application_object, refusal = application()
if application_object is None:
print(refusal)
return None
from .qt.widgets.fast_plots import FastPlot
from .figures.style import ROLES
sizes = summary['sample_size'].to_numpy(dtype=float)
middle = summary['smoothed_mean_abs_diff'].to_numpy(dtype=float)
spread = summary['std_abs_diff'].to_numpy(dtype=float)
plot = FastPlot(title="Mean absolute difference against sample size",
x_label="Sample size",
y_label="Mean absolute difference")
try:
plot.resize(1100, 760)
if not plot.add_curve(sizes, middle, low=middle - spread,
high=middle + spread):
return None
plot.add_line(x=float(mark), colour=ROLES["reference"],
label="minimum cell count")
application_object.processEvents()
return write_plot(plot, path, "Minimum cell count")
except Exception: # noqa: BLE001
plot.deleteLater()
raise
def _figure_name_for(title):
"""A filename from a figure title: lower case, words joined by _."""
keep = [ch.lower() if ch.isalnum() else " " for ch in str(title)]
return "_".join("".join(keep).split()) or "figure"
def _draw_radar_in_pyqtgraph(labels, values, title, src, name,
save=True):
"""Draw a radar in pyqtgraph and write its bundle under ``<src>/results``.
Returns the folder, or None when there is no Qt to render under -- which
`render`'s own refusal explains rather than leaving the run silent.
"""
from .figures.headless import application
from .figure_sink import publish_file
application_object, refusal = application()
if application_object is None:
print(refusal)
return None
from .qt.widgets.fast_plots import FastPlot
from .figures.bundle import save as write_bundle
plot = FastPlot(title=title, x_label="", y_label="")
try:
plot.resize(820, 780)
if not plot.add_radar(labels, values):
return None
application_object.processEvents()
folder = write_bundle(_figure_folder(src, save), name,
render=plot.export,
data=plot.radar_frame(), groups=None,
unit="feature", settings={"figure": name})
finally:
plot.deleteLater()
_announce_the_bundle(folder, title)
return folder
def _draw_importance_in_pyqtgraph(frame, title, src, name, top,
save=True):
"""Draw a ranked importance chart in pyqtgraph and write it out.
The regression and explanation figures were drawn twice -- once in
pyqtgraph for the tab and once in matplotlib for the file -- so one
screen produced two pictures of one number from two code paths that can
disagree. These two were the last that could not move, because twenty
feature names need HORIZONTAL bars and the plot could not draw them.
Returns the bundle folder, or None when there is no Qt to render under.
A None is not a failure: the caller has already written the CSV, and
`render_bundle` says out loud why it could not draw.
:param frame: importance table with ``feature`` and ``importance``.
:param title: the figure's title.
:param src: plate folder; the bundle goes under ``<src>/results``.
:param name: bundle name.
:param top: how many features to show.
:returns: the folder written, or None.
"""
from .figures.headless import application
from .figure_sink import publish_file
application_object, refusal = application()
if application_object is None:
print(refusal)
return None
from .qt.widgets.fast_plots import FastPlot
from .figures.bundle import save as write_bundle
shown = frame.head(int(top))
plot = FastPlot(title=title, x_label="Importance", y_label="")
try:
plot.resize(1200, max(420, 34 * len(shown) + 140))
if not plot.add_ranked_bars(list(shown['feature']),
list(shown['importance']),
highlight=3, descending=False):
return None
application_object.processEvents()
folder = write_bundle(_figure_folder(src, save), name,
render=plot.export,
data=plot.ranked_frame(), groups=None,
unit="feature",
settings={"top_features": int(top),
"figure": name})
finally:
plot.deleteLater()
_announce_the_bundle(folder, title)
return folder
def _save_importance_csv(df, src, filename):
"""Write an importance table to ``<src>/results/<filename>``.
:param df: Importance DataFrame with ``feature`` / ``importance``.
:param src: Plate folder the explained model was scored from.
:param filename: Basename of the CSV to write.
:returns: The full path written.
"""
results_loc = os.path.join(src, 'results')
os.makedirs(results_loc, exist_ok=True)
out_path = os.path.join(results_loc, filename)
df.to_csv(out_path, index=False)
print(f"Saved {out_path}")
return out_path
[docs]
def interpret_vision_model(settings=None):
"""Explain a spacr vision-model score using RF, permutation and SHAP importance, with per-compartment / per-channel radar plots.
Merges per-object measurements from the selected measurement store with a CSV of
predicted scores, runs any combination of RF feature importance,
permutation importance and SHAP over the top features, then
aggregates SHAP contributions into compartment and channel radar
plots so you can see which region (cell / nucleus / pathogen /
cytoplasm) and which fluorescence channel drives the model.
:param settings: Settings dict, canonicalized via
:func:`spacr.settings.set_interpret_vision_model_defaults`.
Key entries:
- ``src`` — folder containing the measurements.
- ``measurement_backend`` / ``measurement_backend_target`` — select
the SQLite, DuckDB, Parquet or PostgreSQL measurement store.
- ``scores`` — CSV of per-object predictions to explain.
- ``score_column`` — column of ``scores`` holding the score.
- ``tables`` — DB tables to merge (default
``['cell','nucleus','pathogen','cytoplasm']``).
- ``feature_importance`` / ``permutation_importance`` / ``shap``
— enable each explainer.
- ``top_features`` — cap on features shown.
- ``nuclei_limit`` / ``pathogen_limit`` — object-count caps.
- ``n_jobs``, ``save``.
:returns: The merged per-object DataFrame — the measurement tables
joined to the scores CSV — that the explainers were fitted on.
Radar and importance plots are rendered, and with ``save=True``
importance CSVs are written under the source folder's ``results/``,
as side effects.
Example:
.. code-block:: python
from spacr.ml import interpret_vision_model
interpret_vision_model({
'src': '/data/plate01',
'scores': '/data/plate01/results/pred.csv',
'score_column': 'pred',
'shap': True, 'top_features': 30,
})
See Also:
:func:`spacr.submodules.interpret_vision_model` — legacy /
alternative entry point returning a dict of importance
DataFrames instead of the merged measurements.
"""
if settings is None:
settings = {}
from .io import (_read_and_merge_data, _report_fan_out, JoinFanOut,
TimelapseKeyMismatch)
from .predictions import crop_name_metadata
from .settings import set_interpret_vision_model_defaults
from .utils import save_settings, _time_column, _measurement_store_for
settings = set_interpret_vision_model_defaults(settings)
save_settings(settings, name='interperate_vision_model', show=True)
def create_extended_radar_plot(values, labels, title):
"""Draw a filled radar for ``values`` labelled by ``labels``.
A RADAR IS A POLYGON, NOT AN AXIS. This was the last figure on the
explanation path that could not move to the screen's renderer, on
the grounds that pyqtgraph has no polar view -- which is true and
was never the obstacle: each label takes an angle, each value a
radius, and `FastPlot.add_radar` draws its own rings because a
radar read against a square grid is unreadable.
"""
return _draw_radar_in_pyqtgraph(
list(labels), list(values), title,
settings['src'], _figure_name_for(title), settings['save'])
def extract_compartment_channel(feature_name):
"""Return ``(compartment, channel)`` parsed from a feature column name."""
compartment = feature_name.split('_')[0]
if compartment == 'cells':
compartment = 'cell'
channels = []
if 'channel_0' in feature_name:
channels.append('channel_0')
if 'channel_1' in feature_name:
channels.append('channel_1')
if 'channel_2' in feature_name:
channels.append('channel_2')
if 'channel_3' in feature_name:
channels.append('channel_3')
if channels:
channel = ' + '.join(channels)
else:
channel = 'morphology'
return (compartment, channel)
def read_and_preprocess_data(settings):
"""Merge measurement DB tables with a scores CSV and split into ``(X, y, merged_df)``."""
sqlite_path = os.path.join(settings['src'], 'measurements',
'measurements.db')
measurement_store = _measurement_store_for(sqlite_path, settings) or sqlite_path
df, _ = _read_and_merge_data(
locs=[measurement_store],
tables=settings['tables'],
verbose=True,
nuclei_limit=settings['nuclei_limit'],
pathogen_limit=settings['pathogen_limit']
)
scores_df = tabular.read_table(settings['scores'])
df['object_label'] = df['object_label'].str.replace('o', '')
join_cols = ['plateID', 'rowID', 'columnID', 'fieldID', 'object_label']
df_time = _time_column(df.columns)
name_col = next((c for c in ('path', 'png_path', 'file_name')
if c in scores_df.columns), None)
if name_col is not None:
parsed = crop_name_metadata(scores_df[name_col],
timelapse=df_time is not None)
for col in parsed.columns:
if col != 'prcfo':
scores_df[col] = parsed[col]
if 'object_label' not in scores_df.columns:
scores_df['object_label'] = scores_df['object']
df['object_label'] = df['object_label'].str.replace('o', '').astype(str)
scores_time = _time_column(scores_df.columns)
if df_time is not None and scores_time is not None:
if df_time != scores_time:
scores_df = scores_df.rename(columns={scores_time: df_time})
join_cols = join_cols + [df_time]
elif df_time is not None or scores_time is not None:
raise TimelapseKeyMismatch(
f"{settings['scores']} and the measurements database disagree "
f"about the timepoint: the scores have {scores_time!r} and the "
f"objects have {df_time!r}. One of the two was produced by a "
f"non-timelapse run, so there is no timepoint to join on, and "
f"joining without it would match every frame's object to every "
f"frame's score. Re-score the dataset, or supply a scores file "
f"that carries the crop file name so the timepoint can be read "
f"off it.")
df[join_cols] = df[join_cols].astype(str)
scores_df[join_cols] = scores_df[join_cols].astype(str)
scores_df = scores_df[join_cols + [settings['score_column']]]
try:
merged_df = pd.merge(df, scores_df, on=join_cols, how='inner',
validate='many_to_one')
except pd.errors.MergeError as error:
duplicated = scores_df[scores_df.duplicated(subset=join_cols,
keep=False)]
examples = (duplicated[join_cols].drop_duplicates()
.head(3).to_dict('records'))
raise JoinFanOut(
f"{settings['scores']} holds more than one score for the same "
f"object: {list(join_cols)} repeats "
f"{len(duplicated[join_cols].drop_duplicates())} time(s), e.g. "
f"{examples}. Joining it to the measurements would put those "
f"objects into the training set once per duplicate row, so "
f"every measurement in the result is duplicated. This usually "
f"means the scoring step ran twice and appended a second set "
f"of rows; de-duplicate the scores file before reading it."
) from error
_report_fan_out(df, merged_df, join_cols,
left_name='object', right_name='scores')
X = schema.model_feature_frame(
merged_df,
exclude=[settings['score_column']],
)
y = merged_df[settings['score_column']]
return X, y, merged_df
X, y, merged_df = read_and_preprocess_data(settings)
if settings['feature_importance'] or settings['permutation_importance'] or settings['shap']:
model = RandomForestClassifier(random_state=_run_random_state(42), n_jobs=settings['n_jobs'])
model.fit(X, y)
feature_importances = model.feature_importances_
feature_importance_df = pd.DataFrame({'feature': X.columns, 'importance': feature_importances})
feature_importance_df = feature_importance_df.sort_values(by='importance', ascending=False)
if settings['feature_importance']:
print(f"Feature Importance ...")
top_feature_importance_df = feature_importance_df.head(settings['top_features'])
_draw_importance_in_pyqtgraph(
feature_importance_df,
f"Top {settings['top_features']} Features - Feature "
f"Importance",
settings['src'], 'feature_importance',
settings['top_features'], settings['save'])
if settings['save']:
_save_importance_csv(feature_importance_df, settings['src'], 'feature_importance.csv')
if settings['permutation_importance']:
print(f"Permutation Importance ...")
perm_importance = permutation_importance(model, X, y, n_repeats=10, random_state=_run_random_state(42), n_jobs=settings['n_jobs'])
perm_importance_df = pd.DataFrame({'feature': X.columns, 'importance': perm_importance.importances_mean})
perm_importance_df = perm_importance_df.sort_values(by='importance', ascending=False)
top_perm_importance_df = perm_importance_df.head(settings['top_features'])
_draw_importance_in_pyqtgraph(
perm_importance_df,
f"Top {settings['top_features']} Features - Permutation "
f"Importance",
settings['src'], 'permutation_importance',
settings['top_features'], settings['save'])
if settings['save']:
_save_importance_csv(perm_importance_df, settings['src'], 'permutation_importance.csv')
if settings['shap']:
import shap
print(f"SHAP Analysis ...")
top_features = feature_importance_df.head(settings['top_features'])['feature']
X_top = X[top_features]
model = RandomForestClassifier(random_state=_run_random_state(42), n_jobs=settings['n_jobs'])
model.fit(X_top, y)
if settings['shap_sample']:
sample = max(1, min(int(len(X_top) / 100), len(X_top)))
X_sample = X_top.sample(sample, random_state=_run_random_state(42))
else:
X_sample = X_top
explainer = shap.Explainer(model.predict, X_sample)
shap_values = explainer(X_sample, max_evals=1500)
_draw_shap_summary_in_pyqtgraph(
shap_values, X_sample, settings['src'], 'shap_summary',
settings['top_features'], settings['save'])
shap_df = pd.DataFrame(shap_values.values, columns=X_sample.columns)
shap_df.columns = pd.MultiIndex.from_tuples(
[extract_compartment_channel(feat) for feat in shap_df.columns],
names=['compartment', 'channel']
)
shap_features = shap_df.abs().T
compartment_mean = (
shap_features.groupby(level='compartment').mean().mean(axis=1))
channel_mean = (
shap_features.groupby(level='channel').mean().mean(axis=1))
combined_compartment = {}
for i, comp1 in enumerate(compartment_mean.index):
for comp2 in compartment_mean.index[i+1:]:
combined_compartment[f"{comp1} + {comp2}"] = shap_df.loc[:, (comp1, slice(None))].abs().mean().mean() + \
shap_df.loc[:, (comp2, slice(None))].abs().mean().mean()
combined_channel = {}
for i, chan1 in enumerate(channel_mean.index):
for chan2 in channel_mean.index[i+1:]:
combined_channel[f"{chan1} + {chan2}"] = shap_df.loc[:, (slice(None), chan1)].abs().mean().mean() + \
shap_df.loc[:, (slice(None), chan2)].abs().mean().mean()
all_compartment_importance = list(compartment_mean.values) + list(combined_compartment.values())
all_compartment_labels = list(compartment_mean.index) + list(combined_compartment.keys())
all_channel_importance = list(channel_mean.values) + list(combined_channel.values())
all_channel_labels = list(channel_mean.index) + list(combined_channel.keys())
create_extended_radar_plot(all_compartment_importance, all_compartment_labels, "SHAP Importance by Compartment (Individual and Combined)")
create_extended_radar_plot(all_channel_importance, all_channel_labels, "SHAP Importance by Channel (Individual and Combined)")
return merged_df
interperate_vision_model = interpret_vision_model