Source code for spacr.refit

"""Re-fit a screen from its plot with a different statistical model.

The regression plot's context menu can rerun an analysis with a different
model, correction method, or FDR threshold.

A re-fit recalculates results rather than changing plot appearance. The menu
therefore presents it separately from styling actions, and
:func:`spacr.ml._next_results_folder` writes the result beside the original
run without overwriting it.

The saved settings require three adjustments before reuse:

  * ``plot`` is forced to False on the way out (:func:`spacr.utils.save_settings`
    does it so a reload reproduces the run headlessly), so re-running from the
    saved copy produces no figures at all, even though the action starts from
    a figure.
  * the per-backend knobs are still set to what the old backend read. Handing
    ``alpha=0.3`` to OLS does not quietly do nothing: :func:`_reject_unused_settings`
    raises, by design, so a lasso -> ols re-fit dies at the entry point.
  * ``regression_type`` and ``random_row_column_effects`` can contradict each
    other, and the reconciliation between them refuses rather than guesses.

:func:`refit_settings` applies these adjustments and reports each change,
including reset backend-specific parameters.
"""

from __future__ import annotations

import os
from typing import Dict, List, Optional, Sequence, Tuple

#: What a re-fit is allowed to change. Deliberately short: this is "the same
#: screen through a different model", not a second settings panel. Anything
#: else -- the data, the formula, the filters -- would make the two runs
#: incomparable, which is the one thing a side-by-side re-fit is for.
REFITTABLE = ("regression_type", "multiple_testing_method", "fdr_alpha",
              "alpha", "cov_type", "quantile", "huber_t", "l1_ratio",
              "spline_knots", "spline_degree",
              "hinge_threshold", "random_row_column_effects")

#: The setting that names the correction. SPELLED THE WAY THE RUN READS IT --
#: :func:`spacr.ml.perform_regression` looks up ``multiple_testing_method``
#: and nothing anywhere reads ``correction_method``, so a re-fit that wrote
#: the second spelling would run with the OLD correction and label its output
#: with the new one. Named here so there is one place to be wrong.
CORRECTION_KEY = "multiple_testing_method"

#: The significance level the correction is applied at. NOT ``alpha``, which
#: is the penalty weight of a penalised fit -- two different numbers, one
#: letter apart, and swapping them silently changes either the hit list or
#: the model.
CORRECTION_ALPHA_KEY = "fdr_alpha"

#: What the run uses when the settings name no level -- `spacr.settings`
#: setdefaults it to this. Restated here only so an absent setting and an
#: explicit 0.05 are not reported as a change from one to the other.
DEFAULT_FDR_ALPHA = 0.05

#: Keys that name where the LAST run wrote. A re-fit must not inherit them:
#: `src` is the run's own output root, and carrying it over would nest the
#: new results inside the old ones. It is rebuilt from the count data, which
#: is the rule the results folder already follows.
RESOLVED_OUTPUT_KEYS = ("results_path", "res_folder", "volcano_path")


def _first_usable_count_path(settings: dict) -> Optional[str]:
    """Return the first non-empty filesystem-compatible count-data path.

    :param settings: Mapping whose ``count_data`` value may be one path or a
        path sequence.
    :returns: The first usable decoded path, or ``None``.
    """
    from .chaining import is_empty_path

    value = settings.get("count_data")
    values = value if isinstance(value, (list, tuple)) else (value,)
    for candidate in values:
        if is_empty_path(candidate):
            continue
        try:
            path = os.fspath(candidate)
        except TypeError:
            continue
        path = os.fsdecode(path).strip()
        if not is_empty_path(path):
            return path
    return None


[docs] def policed_settings() -> Dict[str, object]: """``{setting: the value that means "not asked for"}``. :returns: Backend-specific setting names mapped to their inactive defaults. Values are read from :mod:`spacr.regression_spec`, the same source used by the regression implementation, so reset behavior tracks backend settings. """ from .regression_spec import _MODEL_LEVEL_DEFAULTS, _RUN_LEVEL_DEFAULTS merged = dict(_MODEL_LEVEL_DEFAULTS) merged.update(_RUN_LEVEL_DEFAULTS) return merged
[docs] def prune_for_type(settings: dict, regression_type) -> Tuple[dict, List[str]]: """Reset every knob ``regression_type`` cannot read, and name them. :param settings: settings mapping; not mutated. :param regression_type: the backend about to be fitted. ``None`` means "choose from the data", which reads none of the policed knobs -- so it prunes as strictly as a named type, for the reason :func:`spacr.ml._reject_unused_run_settings` gives. :returns: ``(settings, [what was reset])``. Inapplicable settings are reset rather than deleted so the settings panel retains explicit default values. """ from .regression_spec import REGRESSION_SETTINGS_USED used = REGRESSION_SETTINGS_USED.get(regression_type, ()) out = dict(settings) reset: List[str] = [] for name, default in policed_settings().items(): if name in used or name not in out: continue value = out[name] if name == "alpha" and (value is None or value == "auto"): out[name] = default continue if value == default: continue out[name] = default reset.append(f"{name}={value!r}") return out, reset
[docs] def refit_settings(base: dict, *, regression_type=None, correction_method: Optional[str] = None, fdr_alpha: Optional[float] = None, level: Optional[str] = None, alpha=None) -> Tuple[dict, List[str]]: """Return settings for re-running a screen with a new model. :param base: the settings the run on screen used. :param regression_type: the new backend, or ``None`` to keep the old one. :param correction_method: the new multiple-testing correction, or ``None`` to keep the old one. Written to :data:`CORRECTION_KEY`. :param fdr_alpha: the new significance level, or ``None`` to keep it. :param level: ``'grna'``, ``'gene'`` or ``'both'``, or ``None`` to keep the previous value. Mixed models treat guides as random effects and therefore provide shrunken predictions without per-guide p-values; a guide-level fixed-effect model provides coefficients and p-values. :param alpha: the new penalty weight, where the new backend reads one -- a different number from ``fdr_alpha`` despite the name. :returns: ``(settings, [notes for the user])``. :raises ValueError: if ``base`` names no usable count-data path. Returned notes identify every automatic settings change before execution. """ if not base: raise ValueError( "There are no settings to re-fit from. The panel knows which " "table it is showing but not which settings produced it, which " "happens when results were opened from disk and the run's " "settings CSV is not beside them.") if _first_usable_count_path(base) is None: raise ValueError( "These settings contain no usable count data path, so there is " "nothing to re-fit: a regression needs a count CSV, not just " "the coefficients it produced.") settings = dict(base) notes: List[str] = [] for key in RESOLVED_OUTPUT_KEYS: settings.pop(key, None) if level is not None: from .settings import REGRESSION_LEVELS level = str(level).strip().lower() if level not in REGRESSION_LEVELS: raise ValueError( f"level={level!r} must be one of {list(REGRESSION_LEVELS)}.") old_level = str(settings.get("level", "both")).strip().lower() if level != old_level: settings["level"] = level notes.append(f"level {old_level!r} -> {level!r}") old_type = settings.get("regression_type") if regression_type is not None and regression_type != old_type: settings["regression_type"] = regression_type notes.append(f"model {old_type!r} -> {regression_type!r}") if correction_method is not None: from .multiple_testing import canonical_method correction_method = canonical_method(correction_method) old = settings.get(CORRECTION_KEY) if correction_method != old: settings[CORRECTION_KEY] = correction_method notes.append(f"correction {old!r} -> {correction_method!r}") if fdr_alpha is not None: old = settings.get(CORRECTION_ALPHA_KEY, DEFAULT_FDR_ALPHA) settings[CORRECTION_ALPHA_KEY] = fdr_alpha if fdr_alpha != old: notes.append(f"significance level {old!r} -> {fdr_alpha!r}") chosen = settings.get("regression_type") if alpha is not None: settings["alpha"] = alpha if settings.get("random_row_column_effects") and chosen not in ( None, "mixed"): settings["random_row_column_effects"] = False notes.append("random row/column effects off (they fit a mixed model, " f"and {chosen!r} was asked for)") if (str(settings.get("level", "")).lower() == "grna" and str(chosen or "").lower() == "mixed"): notes.append( "level='grna' with regression_type='mixed' still gives BLUPs " "and no per-guide p value, because a mixed model makes the " "guide a random effect. For per-guide significance choose a " "fixed-effect model such as 'ols', 'rlm' or 'quantile', or use " "inference='nonparametric'.") settings, reset = prune_for_type(settings, chosen) if reset: notes.append(f"{chosen!r} does not read " + ", ".join(reset) + " — reset to default") if settings.get("plot") is False: settings["plot"] = True settings["test_mode"] = False settings.pop("src", None) return settings, notes
[docs] def destination(settings: dict) -> Optional[str]: """Return the output directory a run would use without executing it. :param settings: regression settings used to resolve source and model kind. :returns: The predicted results directory, or ``None`` when no count-data path or writable folder choice can be resolved. PostgreSQL counts require an explicit local ``src``. The result uses the same folder-selection rule as the regression run. """ from .ml import _next_results_folder count = _first_usable_count_path(settings) if count is None: return None from . import tabular requested = settings.get("src") if tabular._backend_of(count) == "postgres" and str(requested or "").strip() in ( "", "path", "/path", "/path/to/src"): return None if requested and tabular._backend_of(requested) == "postgres": return None src = requested or os.path.dirname(str(count)) kind = ("guide_permutation" if settings.get("analysis_mode") == "guide_permutation" else settings.get("regression_type") or "auto") try: return _next_results_folder(os.path.join(src, "results"), str(kind)) except OSError: return None
#: Where a run leaves the settings it actually used, relative to the folder #: holding its results table. `save_settings` writes under the run's `src`, #: which is the results ROOT rather than the run's own folder -- so a re-fit #: started from an old table has to look upwards, and may find a settings #: file from a LATER run. Named here so the search order is one decision in #: one place. SETTINGS_NAMES = ("regression_settings.csv", "regression.csv", "settings.csv")
[docs] def settings_of_run(results_path) -> Optional[dict]: """The settings a finished run used, read back from beside its results. :param results_path: the results CSV, or the folder holding it. :returns: the parsed settings, or ``None`` when the run left none. Search proceeds from the run's own folder to ``settings/`` beside the results root. This prioritizes run-specific settings over the shared copy, which later runs may overwrite. """ from .utils import load_settings if results_path is None: return None folder = str(results_path) if os.path.isfile(folder): folder = os.path.dirname(folder) roots: Sequence[str] = (folder, os.path.dirname(folder), os.path.dirname(os.path.dirname(folder))) for root in roots: if not root: continue for name in SETTINGS_NAMES: for candidate in (os.path.join(root, name), os.path.join(root, "settings", name)): if not os.path.isfile(candidate): continue try: return load_settings(candidate) except Exception: # noqa: BLE001 continue return None
__all__ = ["CORRECTION_ALPHA_KEY", "CORRECTION_KEY", "DEFAULT_FDR_ALPHA", "REFITTABLE", "RESOLVED_OUTPUT_KEYS", "SETTINGS_NAMES", "destination", "policed_settings", "prune_for_type", "refit_settings", "settings_of_run"]