"""Inspect coefficient tables and diagnostics from a finished regression.
The results panel combines a clickable volcano plot, sortable coefficient
table, p-value histogram, Q-Q calibration view, assay controls, guide-support
view, model summary, and annotation for the selected gene. Selecting a
coefficient in any linked plot or table highlights the same feature in the
other views.
Which coefficients every tab draws -- gRNA, Gene or Both -- is the reader's
choice, made on the panel beside what it changes. A run fitted at
``level='both'`` fits twice and writes both families into one table, so the
gene half needs no re-fit to be seen. The two fits are separate
multiple-testing families and are corrected as such, and each row says which
fit it came from, so a gene and its guides are distinct rows in the table, on
the plot and in the exported copy. A level this run has no rows at says why it
has none rather than drawing an empty tab.
Gene and gRNA filters are applied to coefficient-level views, including their
calibration diagnostics. Well-level residual and influence diagnostics remain
unfiltered because wells do not belong to either coefficient family. Plot
selections are joined by feature keys rather than drawing positions, so
sorting or aggregation does not change which result is selected.
"""
from __future__ import annotations
import logging
import os
import re
from typing import Dict, Optional
from PySide6.QtCore import Qt, Signal
from PySide6.QtWidgets import (
QComboBox, QHBoxLayout, QLabel, QPushButton, QTabWidget,
QVBoxLayout, QWidget,
)
from ..i18n import tr
LOG = logging.getLogger(__name__)
#: How the effect-size cut is measured, and how wide, on a run nobody has
#: told otherwise. Named so :meth:`RegressionResultsPanel.set_frame`'s reset
#: and the constructor cannot drift: they used to be two literals, and the
#: reset did not exist at all -- so run B inherited run A's cut.
DEFAULT_THRESHOLD_METHOD = "mad"
DEFAULT_THRESHOLD_MULTIPLIER = 3.0
#: Files a regression writes, best first. ``results.csv`` is the full
#: coefficient table; the gene/grna splits are views of it.
#:
#: ``guide_permutation_results_long.csv`` comes LAST, and the position is the
#: point. A guide permutation stacks every ``minimum_wells_threshold`` and
#: every ``outcome`` into that one table, so it holds several multiple-testing
#: families at once and only the primary slice of it reaches ``results.csv``
#: beside it. Naming it here is what makes a run folder that holds nothing
#: else loadable at all -- before, "Load results…" searched such a folder and
#: reported finding none of these names. Naming it last is what keeps the
#: single-family ``results.csv`` the table that opens whenever both are there,
#: because :func:`find_results_tables` loads the first match and lists the
#: rest.
RESULT_FILENAMES = ("results.csv", "results_grna.csv", "results_gene.csv",
"guide_permutation_results_long.csv")
#: How far below the selected folder to search for a results table. Current
#: runs write ``results/<kind>[_n]/results.csv``; the additional depth keeps
#: older layouts discoverable without walking an entire home directory.
MAX_SEARCH_DEPTH = 6
#: Stop after this many candidates. Only the newest is shown and the rest are
#: a count in the status line, so there is nothing to gain from finding a
#: thousand of them.
MAX_CANDIDATES = 200
#: Every spelling of the p-value column this panel can be handed, best first.
#:
#: spaCR's own ``process_model_coefficients`` writes ``p_value``. A table that
#: came through a statsmodels summary carries ``P>|t|`` for OLS and ``P>|z|``
#: for a GLM or a Poisson fit -- the ``t value``/``z value`` beside it is the
#: STATISTIC, not the p -- and an R-shaped export spells it ``Pr(>|t|)``.
#: Matching is done on a normalised form, so case and separators do not count.
P_VALUE_COLUMNS = (
"p_value", "p", "pvalue", "p-value", "p_val", "pval",
"P>|t|", "P>|z|", "Pr(>|t|)", "Pr(>|z|)",
)
#: The ranking the penalised backends report instead. ``lasso`` and
#: ``elasticnet`` have no frequentist p-value at all -- see
#: :data:`spacr.hits.NO_P_VALUE_TYPES` -- so a bootstrap selection frequency
#: is what orders their coefficients.
SELECTION_COLUMNS = ("selection_frequency", "selection_freq",
"proportion_selected")
#: Columns only a guide permutation writes. A permutation resamples the well
#: labels of ONE guide at a time, so its table is per guide by construction
#: and holds no gene-level test at all -- which is a fact about the inference
#: rather than a fault in the run, and is the reason "Gene" can be an honest
#: empty on a run that finished perfectly. See
#: :meth:`RegressionResultsPanel.missing_level_note`.
PERMUTATION_COLUMNS = ("permutation_p_value", "permutation_exceedances",
"permutations")
#: Test statistics. Present without a p-value they are worth naming in the
#: status line, because "z value" is the thing a user will point at and ask
#: why the histogram is empty.
STATISTIC_COLUMNS = ("z value", "t value", "z_value", "t_value", "z", "t",
"statistic", "wald")
def _normalise(name) -> str:
"""A column name reduced to what actually distinguishes it."""
return "".join(str(name).lower().split()).replace("_", "").replace("-", "")
def _match_column(frame, wanted) -> Optional[str]:
"""The first column of ``frame`` matching one of ``wanted``, in order."""
if frame is None:
return None
columns = list(getattr(frame, "columns", ()))
for name in wanted:
if name in columns:
return name
lookup = {}
for column in columns:
lookup.setdefault(_normalise(column), column)
for name in wanted:
found = lookup.get(_normalise(name))
if found is not None:
return found
return None
[docs]
def find_results_tables(path, *, max_depth: int = MAX_SEARCH_DEPTH,
limit: int = MAX_CANDIDATES) -> list:
"""Find regression result tables beneath a path.
Parameters
----------
path : path-like
Results CSV, run directory, or parent directory to search. A CSV is
returned directly; missing paths and non-CSV files return no matches.
max_depth : int, default=MAX_SEARCH_DEPTH
Maximum directory depth to descend below ``path``. The root directory
is still inspected when this value is zero.
limit : int, default=MAX_CANDIDATES
Stop walking after at least this many candidate tables have been
collected. All recognised tables in the final run directory are kept,
so the returned count can be slightly larger than this value.
Returns
-------
list of str
Recognised result-table paths. Run directories are ordered by the
newest modification time among their tables. Within a run,
``results.csv`` precedes the gene and gRNA views according to
:data:`RESULT_FILENAMES`.
"""
if not path:
return []
try:
root = os.path.abspath(os.path.expanduser(os.fspath(path)))
except TypeError:
return []
if os.path.isfile(root):
if not root.lower().endswith(".csv"):
return []
home = os.path.dirname(root)
rest = []
for name in RESULT_FILENAMES:
beside = os.path.join(home, name)
if beside != root and os.path.isfile(beside):
rest.append(beside)
return [root] + rest
if not os.path.isdir(root):
return []
runs = []
total = 0
for folder, subdirs, files in os.walk(root, followlinks=False):
depth = folder[len(root):].count(os.sep)
if depth >= max_depth:
subdirs[:] = []
subdirs.sort()
present = set(files)
found = []
newest = None
for order, name in enumerate(RESULT_FILENAMES):
if name not in present:
continue
candidate = os.path.join(folder, name)
try:
stamp = os.path.getmtime(candidate)
except OSError:
continue
found.append((order, candidate))
newest = stamp if newest is None else max(newest, stamp)
if found:
found.sort()
runs.append((-newest, folder, [c for _, c in found]))
total += len(found)
if total >= limit:
break
runs.sort()
return [candidate for _, _, group in runs for candidate in group]
[docs]
def read_run_tables(tables, progress=None):
"""Read a run's primary table, and any LEVEL it left in a sibling file.
``results.csv`` is meant to hold every level a run produced -- a fitted
run at ``level='both'`` writes its guide and gene rows into one table and
the panel filters them apart by the ``level`` column.
THE PERMUTATION PATH DID NOT ALWAYS DO THAT. It tested genes, corrected
them as their own family, wrote them to ``results_gene.csv``, and left
``results.csv`` holding guides alone -- so a reader who asked for genes
was shown nothing while the rows sat in a file nothing opened. The writer
is fixed, and this reads the older runs too, because a folder on disk
does not update itself.
A sibling is merged only when it brings a level the primary table does
NOT already carry, so a run that already holds both is untouched and
nothing is ever counted twice.
:param tables: candidate paths, primary first, as
:func:`find_results_tables` returns them.
:param progress: optional ``progress(done, total, name)``, called before
each file is read. ``total`` counts the primary table and every
sibling beside it, so it is an upper bound: a sibling whose level the
primary already carries is counted and not read.
:returns: ``(frame, found, merged)`` -- the table, the path it came from,
and the sibling paths folded into it.
"""
import pandas as pd
from ...tabular import read_table
found = tables[0]
home = os.path.dirname(found)
total = 1 + sum(1 for sibling in tables[1:]
if os.path.dirname(sibling) == home)
if progress is not None:
progress(1, total, os.path.basename(found))
frame = read_table(found, report=None)
merged = []
have = set()
if "level" in frame.columns:
have = {str(v).strip().lower() for v in frame["level"].dropna().unique()}
else:
from spacr.hits import coefficient_levels
try:
have = {str(v) for v in pd.Series(coefficient_levels(frame)).unique()}
except Exception: # noqa: BLE001
have = set()
done = 1
for sibling in tables[1:]:
if os.path.dirname(sibling) != home:
continue
done += 1
name = os.path.basename(sibling)
wants = ("gene" if "gene" in name else
"grna" if "grna" in name else "")
if not wants or wants in have:
continue
if progress is not None:
progress(done, total, name)
try:
extra = read_table(sibling, report=None)
except Exception: # noqa: BLE001
continue
if extra.empty:
continue
if "level" not in extra.columns:
extra = extra.copy()
extra["level"] = wants
if "level" not in frame.columns:
frame = frame.copy()
frame["level"] = sorted(have)[0] if len(have) == 1 else None
frame = pd.concat([frame, extra], ignore_index=True, sort=False)
have.add(wants)
merged.append(sibling)
return frame, found, merged
#: Columns the coefficient table keeps even when every visible row leaves
#: them blank, because a reader looks for them by name and an absent column
#: reads as a missing measurement rather than an empty one.
TABLE_KEEP_COLUMNS = ("feature", "level", "coefficient", "p_value",
"q_value", "adjusted_p_value", "significant")
[docs]
def for_table(frame):
"""``frame`` without the columns that are blank for every row shown.
A permutation run's table is the UNION of two schemas: a guide row has
``guide`` and ``wells_with_guide``, a gene row has ``gene``,
``wells_with_gene`` and ``guides_in_gene``. Showing the union means that
whichever level is chosen, a third of the columns are empty -- and the
gene the reader is looking for is named thirteen columns to the right of
a blank ``guide`` cell, which is what makes a gene view read as broken
even once the rows are there.
Dropping a column that no visible row fills is safe because it carries no
value for this selection: "how many wells hold this guide" has no answer
for a gene. The identity and significance columns in
:data:`TABLE_KEEP_COLUMNS` stay regardless.
:param frame: the rows about to be shown.
:returns: the same rows, narrowed. The input is not modified.
"""
columns = list(getattr(frame, "columns", ()))
if not columns or len(frame) == 0:
return frame
if ("coefficient" in columns
and "standardized_marginal_effect" in columns):
try:
same = frame["coefficient"].equals(
frame["standardized_marginal_effect"])
except Exception: # noqa: BLE001
same = False
if same:
columns = [c for c in columns if c != "coefficient"]
frame = frame[columns]
keep = []
for name in columns:
if name in TABLE_KEEP_COLUMNS:
keep.append(name)
continue
column = frame[name]
try:
filled = column.notna()
if filled.any() and column.astype(str).str.strip().ne("").any():
keep.append(name)
except Exception: # noqa: BLE001
keep.append(name)
if len(keep) == len(columns):
return frame
return frame[keep]
[docs]
def find_results_table(path) -> Optional[str]:
"""The results CSV for ``path``: a file, a run folder, or a parent of one.
Those are the three things a user actually has to hand when they want to
look at a regression again, so all three are accepted. Where a parent
holds several, the most recently modified wins -- see
:func:`find_results_tables`.
:param path: results CSV, run folder or a folder above one.
"""
tables = find_results_tables(path)
return tables[0] if tables else None
def _summary_filenames() -> tuple:
"""Every name a run's summary may be on disk under, newest first.
Read from :mod:`spacr.ml`, which is the module that WRITES it -- one
vocabulary, so a rename there cannot leave this reader hunting for a file
nobody makes any more. Private because it is an implementation detail of
:func:`find_summary_file`, which is the thing a caller wants.
"""
try:
from ...ml import SUMMARY_FILENAMES
except Exception:
return ()
return tuple(SUMMARY_FILENAMES)
[docs]
def find_summary_file(path) -> Optional[str]:
"""Find the model summary associated with a regression result.
Parameters
----------
path : path-like
Results CSV, run directory, or parent directory containing a
discoverable results table.
Returns
-------
str or None
Path to the first supported summary file, or ``None`` when no summary
is found. Both the current and legacy summary filenames are accepted.
"""
names = _summary_filenames()
if not path or not names:
return None
try:
root = os.path.abspath(os.path.expanduser(os.fspath(path)))
except TypeError:
return None
folders = []
if os.path.isfile(root):
folders.append(os.path.dirname(root))
elif os.path.isdir(root):
folders.append(root)
table = find_results_table(root)
if table:
folder = os.path.dirname(table)
if folder not in folders:
folders.append(folder)
for folder in folders:
for name in names:
candidate = os.path.join(folder, name)
if os.path.isfile(candidate):
return candidate
return None
#: What the run says when a fit is not identifiable. Repeated on the Summary
#: tab because statsmodels prints a full table of standard errors and P
#: values regardless, and it looks exactly like a summary of a well-posed
#: fit -- which a reader may paste into a methods section from one click
#: away.
UNIDENTIFIABLE_WARNING = (
"THIS FIT IS SATURATED OR NOT IDENTIFIABLE: {wells} analysed "
"observations are being "
"used to estimate {params} parameters.\n"
"With no residual degrees of freedom, and possible rank deficiency, "
"individual coefficients, standard errors and P values cannot be "
"interpreted reliably.\n"
"Set inference='nonparametric' to test each guide as a marginal "
"association, reshuffling wells only between wells of the same plate, "
"without fitting every guide coefficient at once; or use "
"inference='auto' to let spaCR choose.\n")
#: Prefixed to a summary that was READ rather than rendered. A reader pasting
#: this into a methods section is entitled to know it is the run's own text,
#: recovered from the run folder and not recomputed here -- and if the folder
#: were ever pointed at the wrong run, this line is what would show it.
#: The first line of a summary spaCR wrote itself, and the section heading it
#: files the statsmodels text under. Both spelled by
#: :func:`spacr.regression_summary.format_run_summary`; matched rather than
#: re-derived, so a rename there shows up as the spaCR summary going missing
#: from this tab rather than as a wrong-looking one.
SPACR_SUMMARY_HEADING = "spaCR RUN SUMMARY"
VERBATIM_HEADING = "THE STATSMODELS SUMMARY"
SUMMARY_FROM_DISK = (
"Read from {path}: this is the run's own summary, written when it was "
"fitted, not recomputed here.")
#: Why there is no fitted model, when nothing more specific is known. Used
#: ONLY where it has been checked -- see :func:`summary_text`.
NO_MODEL_FROM_DISK = (
"this panel was opened from a results table on disk rather than from a "
"run, so the fitted model is not here")
NO_MODEL_AT_ALL = "no run in this session has fitted anything yet"
_CANONICAL_TABLE = "results.csv"
def _why_no_model(path, reason) -> str:
"""Return the verified reason that no fitted model is available.
Prefer an explicit reason. Otherwise distinguish a table loaded from disk
from a session in which no model has been fitted.
"""
if reason:
return str(reason)
return NO_MODEL_FROM_DISK if path else NO_MODEL_AT_ALL
[docs]
def summary_text(model, regression_type=None, *, path=None,
reason: str = "") -> str:
"""Return a model summary or an explanation of its absence.
Parameters
----------
model : object or None
Fitted model. Objects with a callable ``summary`` method use that
method; ``None`` triggers lookup of a summary saved with the run.
regression_type : str or None, optional
Backend name included when the fitted object has no statsmodels-style
summary.
path : path-like or None, optional
Results CSV, run directory, or parent directory. When ``model`` is
``None``, the function looks beside the selected results table for a
summary written during fitting.
reason : str, optional
Known reason that no live model is available. When omitted, the
explanation is limited to what can be inferred from ``path``.
Returns
-------
str
Saved or live summary text. If no summary is available, a message
names the relevant backend, path, or read failure instead of returning
an empty string.
Notes
-----
A live statsmodels summary is returned from the fitted model rather than
reconstructed. When the run summary is available on disk, it is included
with the model summary.
"""
if model is None:
found = find_summary_file(path)
if found:
try:
with open(found, "r", encoding="utf-8", errors="replace") as f:
text = f.read().strip()
except OSError as error:
return (f"No summary: {_why_no_model(path, reason)}, and the "
f"summary this run wrote could not be read "
f"({error}).")
if text:
return f"{SUMMARY_FROM_DISK.format(path=found)}\n\n{text}"
return (f"No summary: {_why_no_model(path, reason)}, and the "
f"summary file this run wrote ({found}) is empty.")
why = _why_no_model(path, reason)
if path:
names = _summary_filenames()
looked = os.path.dirname(str(path)) if os.path.isfile(
str(path)) else str(path)
return (f"No summary: {why}, and this run wrote none — looked "
f"for {' or '.join(names) if names else 'a summary file'} "
f"in {looked}. Re-running with this build writes one.")
return f"No summary: {why}."
summary = getattr(model, "summary", None)
if not callable(summary):
named = f" ({regression_type})" if regression_type else ""
return _with_spacr_summary(
path,
f"this backend{named} is not a statsmodels fit, so it has none. "
f"Lasso, elastic net, and group lasso are ranked by bootstrap "
f"selection frequency because they do not report frequentist "
f"p-values. Ridge and hinge uncertainty is reported in the "
f"Coefficients tab.",
missing=True)
try:
text = str(summary())
except Exception as error: # noqa: BLE001
return _with_spacr_summary(
path,
f"statsmodels could not render one for this fit "
f"({type(error).__name__}: {error}).",
missing=True)
warning = _identifiability_warning(model)
return _with_spacr_summary(path, f"{warning}\n{text}" if warning else text)
def _with_spacr_summary(path, statsmodels_text: str, *,
missing: bool = False) -> str:
"""Prepend the spaCR run summary to the statsmodels summary.
The saved summary's embedded statsmodels tail is removed so the live text
is not duplicated. When ``missing`` is true, the returned second section
explains why no statsmodels summary is available.
"""
spacr = _spacr_summary_text(path)
if not spacr:
return f"No summary: {statsmodels_text}" if missing else statsmodels_text
body = (f"No statsmodels summary: {statsmodels_text}" if missing
else statsmodels_text)
return f"{spacr}\n\n{VERBATIM_HEADING}\n{body}"
def _spacr_summary_text(path) -> str:
"""spaCR's own summary for a run, without its verbatim tail. "" if none.
The file already carries the statsmodels text as its last section (see
`write_run_summary`), so that tail is trimmed rather than printed twice --
the live fit is the one on screen and is the one to keep.
"""
found = find_summary_file(path)
if not found:
return ""
try:
with open(found, "r", encoding="utf-8", errors="replace") as handle:
text = handle.read().strip()
except OSError:
return ""
if not text or not text.startswith(SPACR_SUMMARY_HEADING):
return ""
cut = text.find(VERBATIM_HEADING)
return text[:cut].rstrip() if cut > 0 else text
def _identifiability_warning(model) -> str:
"""The run's own not-identifiable warning, or "".
Read off the model rather than off the settings, so the tab cannot
disagree with the table it is printed above.
"""
try:
observations = int(getattr(model, "nobs", 0))
params = len(getattr(model, "params", ()))
except Exception: # noqa: BLE001
return ""
if observations and params and params >= observations:
return UNIDENTIFIABLE_WARNING.format(wells=observations,
params=params)
return ""
[docs]
def backend_of(path) -> Optional[str]:
"""The regression type a results path was written under, if it says.
``perform_regression`` writes to ``results/<kind>[_n]/``. The folder name
is therefore the only evidence available when a table does not record its
backend directly.
:param path: results file or folder path, or ``None``. A path component
that, lower-cased and without a trailing ``_<n>``, names one of
``spacr.hits.NO_P_VALUE_TYPES`` is returned; otherwise ``None``.
"""
if not path:
return None
try:
from ...hits import NO_P_VALUE_TYPES
except Exception:
return None
parts = {
re.sub(r"_\d+$", "", part.strip().lower())
for part in str(path).replace("\\", "/").split("/")
}
for name in NO_P_VALUE_TYPES:
if name in parts:
return name
return None
[docs]
class RegressionResultsPanel(QWidget):
"""Volcano, table and diagnostics for one finished regression."""
#: Emitted with the results CSV whenever a new one is loaded.
loaded = Signal(str)
#: An asynchronous load finished. ``True`` when the run is on screen.
#: Separate from :attr:`loaded`, which carries a path and predates the
#: worker: a caller needs to know SUCCESS, and it arrives later than the
#: call that started it.
load_finished = Signal(bool)
#: What the run label says when the table on screen came from no run --
#: a bare CSV, or a frame handed straight in. NOT blank: an empty label
#: reads as "the run has no name" rather than "this table has no run".
NO_RUN_NAMED = "No run"
#: Emitted with a settings dict when the user asks, from the plot, for
#: the same screen through a different model. The panel does not start
#: the run itself -- it has no worker, no console and no Stop button --
#: so whichever screen owns those does.
refit_requested = Signal(object)
_load_progress_relayed = Signal(int, int, int, int, str)
LOAD_STEPS = 3
def __init__(self, parent=None, external_volcano: bool = False):
"""Initialize the regression results panel.
Parameters
----------
parent : QWidget or None, optional
Parent widget.
external_volcano : bool, default=False
Build and wire the interactive volcano plot without adding it to
this panel's layout. Use this when a host places the volcano in a
larger external view; selection and redraw behavior are unchanged.
"""
super().__init__(parent)
from .fast_plots import (ControlSeparation, EffectDistribution,
EffectRankPlot, GuideAgreementPlot,
InfluencePlot, PValueHistogram, QQPlot,
ResidualPlot, ResultsTable,
ScaleLocationPlot, VolcanoPlot)
from .flow import FlowHost, FlowLayout
layout = QVBoxLayout(self)
layout.setContentsMargins(0, 0, 0, 0)
layout.setSpacing(4)
header = QHBoxLayout()
self._load_button = QPushButton("Load results…")
self._load_button.setToolTip(
"Choose a regression results folder, or a parent of one. The most "
"recently written results table under it is loaded.")
self._load_button.clicked.connect(self.browse_for_results)
header.addWidget(self._load_button)
self._source = QLabel("No regression loaded.")
self._source.setWordWrap(True)
header.addWidget(self._source, 1)
controls_row = FlowHost()
controls = FlowLayout(controls_row, spacing=6)
#: The wrapping row itself, named so its HEIGHT can be measured: a
#: row that wrapped and a row that overlapped both fit the panel, and
#: only the height tells them apart.
self._controls_row = controls_row
self._run_label = QLabel(self.NO_RUN_NAMED)
self._run_label.setObjectName("resultsRunName")
controls.addWidget(self._run_label)
self._level_label = QLabel("level")
self._level_label.setToolTip(
"Which family of coefficients every tab draws.\n\n"
"A run fitted at level='both' fits TWICE — once per guide and "
"once per gene — and writes both into one table. They are two "
"multiple-testing families, so the q-values, the Q-Q's inflation "
"figure and the significance line are computed within each "
"family and not across the pair.\n\n"
"Both shows them together. A gene and its guides are then "
"separate rows with separate labels: gene_fraction:gene[225160] "
"beside fraction:grna[225160_1], and a level column in the "
"table that says which is which.")
controls.addWidget(self._level_label)
self._level_box = QComboBox()
self._level_box.setMinimumWidth(120)
self._level_box.setToolTip(self._level_label.toolTip())
self._level_box.activated.connect(self._level_chosen)
controls.addWidget(self._level_box)
self._colour_by_label = QLabel("colour by")
self._colour_by_label.setToolTip(
"Up to three columns can be encoded simultaneously: the "
"first is the colour, the second the marker shape, the third the "
"opacity. The order is fixed so each position has a consistent "
"visual meaning. Additional encodings are not supported because "
"they would reduce interpretability.\n\n"
"Choosing nothing in the first box turns the other two off as "
"well, preventing shape or opacity from being displayed without "
"the primary colour encoding.")
controls.addWidget(self._colour_by_label)
self._colour_by = QComboBox()
self._colour_by.setMinimumWidth(140)
self._colour_by.currentIndexChanged.connect(self._redraw_volcano)
controls.addWidget(self._colour_by)
self._colour_by_2 = QComboBox()
self._colour_by_2.setMinimumWidth(120)
self._colour_by_2.setToolTip(
"Second column, encoded as marker shape rather than colour. The "
"encodings are layered rather than combined to avoid a rapidly "
"expanding legend.")
self._colour_by_2.currentIndexChanged.connect(self._redraw_volcano)
controls.addWidget(self._colour_by_2)
self._colour_by_3 = QComboBox()
self._colour_by_3.setMinimumWidth(120)
self._colour_by_3.setToolTip(
"Third column, encoded as opacity. No additional column is "
"offered because more simultaneous encodings reduce readability.")
self._colour_by_3.currentIndexChanged.connect(self._redraw_volcano)
controls.addWidget(self._colour_by_3)
layout.addLayout(header)
layout.addWidget(controls_row)
self._missing_level = QLabel("")
self._missing_level.setObjectName("Muted")
self._missing_level.setWordWrap(True)
self._missing_level.setTextInteractionFlags(Qt.TextSelectableByMouse)
self._missing_level.setVisible(False)
layout.addWidget(self._missing_level)
self.tabs = QTabWidget()
layout.addWidget(self.tabs, 1)
self.volcano = VolcanoPlot()
self.table = ResultsTable()
self.external_volcano = bool(external_volcano)
self.table.setMinimumHeight(150)
self.table.table.setContextMenuPolicy(Qt.CustomContextMenu)
self.table.table.customContextMenuRequested.connect(self._level_menu_at)
if self.external_volcano:
self._volcano_tab, self._volcano_tab_name = self.table, "Coefficients"
self.tabs.addTab(self.table, "Coefficients")
else:
from .collapsible_splitter import CollapsibleSplitter
split = CollapsibleSplitter(Qt.Vertical,
persist_key="regression::volcano")
split.add_section(self.volcano, "Volcano plot", stretch=3,
extent=340,
persist_key="regression/Volcano plot")
self._model_line = QLabel("")
self._model_line.setObjectName("Muted")
self._model_line.setWordWrap(True)
self._model_line.setVisible(False)
split.add_pane(self._model_line, "Model", stretch=0)
split.add_section(self.table, "Coefficient table", stretch=2,
extent=220,
persist_key="regression/Coefficient table")
self.volcano.setMinimumHeight(240)
self._volcano_tab, self._volcano_tab_name = split, "Volcano"
self.tabs.addTab(split, "Volcano")
self.effect_rank = EffectRankPlot()
self.effect_distribution = EffectDistribution()
self.tabs.addTab(self.effect_rank, "Effect rank")
self.tabs.addTab(self.effect_distribution, "Effect distribution")
self.tabs.setTabToolTip(
self.tabs.indexOf(self.effect_rank),
"Every coefficient ranked by the size of its effect, as a dot "
"with its confidence interval. The volcano ranks by significance, "
"which is a different order.")
self.tabs.setTabToolTip(
self.tabs.indexOf(self.effect_distribution),
"Where this screen's effects sit, and how wide the null under "
"them is. σ is a MAD, which the outliers a screen is looking for "
"do not inflate.")
self.p_values = PValueHistogram()
self.qq = QQPlot()
self.controls = ControlSeparation()
self.tabs.addTab(self.p_values, "p-values")
self.tabs.addTab(self.qq, "Q-Q")
self.tabs.addTab(self.controls, "Controls")
from PySide6.QtGui import QFontDatabase
from .folding_summary import FoldingSummaryView
self._summary = FoldingSummaryView()
self._summary.setPlainText(
"Run a regression to see its summary.")
self.residuals = ResidualPlot()
self.scale_location = ScaleLocationPlot()
self.influence = InfluencePlot()
self.tabs.addTab(self.residuals, "Residuals")
self.homogeneity = QLabel(self.NO_HOMOGENEITY_VERDICT)
self.homogeneity.setWordWrap(True)
self.homogeneity.setTextInteractionFlags(Qt.TextSelectableByMouse)
self._scale_location_tab = QWidget()
spread = QVBoxLayout(self._scale_location_tab)
spread.setContentsMargins(0, 0, 0, 0)
spread.setSpacing(4)
spread.addWidget(self.scale_location, 1)
spread.addWidget(self.homogeneity)
self.tabs.addTab(self._scale_location_tab, "Scale-location")
self.tabs.addTab(self.influence, "Influence")
try:
from .annotation_umap_tab import AnnotationUmapTab
self.annotation_umap = AnnotationUmapTab(self)
self.tabs.addTab(self.annotation_umap, "Annotation check")
except Exception: # noqa: BLE001
self.annotation_umap = None
self.tabs.addTab(self._summary, "Summary")
self.tabs.setTabToolTip(
self.tabs.indexOf(self.residuals),
"Residual against fitted, one point per well. A horizontal band "
"is a well-specified mean; a curve or a funnel is not.")
self.tabs.setTabToolTip(
self.tabs.indexOf(self._scale_location_tab),
"The spread of the residuals with the sign taken out. A rising "
"trend means the standard errors, and so every p-value on the "
"volcano, depend on the fitted value.")
self.tabs.setTabToolTip(
self.tabs.indexOf(self.influence),
"Leverage against standardised residual. Which wells are moving "
"the coefficients on their own.")
self.agreement = GuideAgreementPlot()
self.support = ResultsTable()
from .collapsible_splitter import CollapsibleSplitter
agreement_split = CollapsibleSplitter(
Qt.Vertical, persist_key="regression::guide_support")
agreement_split.add_section(
self.agreement, "Guide agreement", stretch=3, extent=340,
persist_key="regression/Guide agreement")
agreement_split.add_section(
self.support, "Guide support table", stretch=2, extent=220,
persist_key="regression/Guide support table")
self.agreement.setMinimumHeight(240)
self.support.setMinimumHeight(150)
self._support_tab = agreement_split
self.tabs.addTab(agreement_split, "Guide support")
from .gene_panel import GenePanel
self.gene = GenePanel(frame_provider=lambda: self._frame)
if not self.external_volcano:
self.tabs.addTab(self.gene, "Gene")
for plot in self._keyed_plots():
plot.key_selected.connect(self._select_from_a_plot)
self.p_values.key_selected.connect(self.table.select_key)
self.p_values.keys_selected.connect(self._show_keys)
self.effect_distribution.key_selected.connect(self.table.select_key)
self.effect_distribution.keys_selected.connect(self._show_keys)
self.table.key_selected.connect(self._select_key)
self.table.key_selected.connect(self.gene.show_feature)
for plot in self._keyed_plots():
if plot in (self.p_values, self.effect_distribution):
continue
plot.keys_selected.connect(self._select_many_from_a_plot)
self._frame = None
self._path = None
self._selected_key = None
#: Per-run plot state, keyed by the run's source path.
#:
#: Only the visible run keeps a live volcano widget; other runs retain
#: lightweight state so switching runs does not multiply redraw cost.
self._plot_states: Dict[str, dict] = {}
self._status = "No regression loaded."
self._ranking = (None, None)
#: Settings that produced the table on screen, when known.
self._run_settings = None
#: ``(kind, name)`` -- what effects are measured FROM. Born here for
#: the same reason as everything else in this block: `_redraw_volcano`
#: reads it and is reachable from a signal connected above.
self._baseline = (None, None)
#: The one TAGM/LOPIT compartment picked out against grey, or None.
self._compartment = None
#: "raw" or "adjusted" -- which p-value the volcano's y-axis is.
self._p_value_kind = "raw"
self._p_value_note = ""
#: How the effect-size cut is measured, and how wide.
self._threshold_method = DEFAULT_THRESHOLD_METHOD
self._threshold_multiplier = DEFAULT_THRESHOLD_MULTIPLIER
#: Why a colour-by option is present but useless, or "".
self._colour_by_note = ""
from ..job_runner import JobRunner
self._loading = False
self._load_generation = 0
self._load_jobs = JobRunner(self, threaded=True, app_key="results")
self._load_jobs.job_failed.connect(self._on_load_job_failed)
self._load_progress_relayed.connect(self._on_load_progress)
#: None, "gene" or "grna" -- which rows EVERY tab draws. One piece of
#: state, read by every draw path: see :meth:`refresh_views`.
self._level = self._default_level()
#: Fitted model behind the table when the current process produced it.
self._model = None
#: What went wrong drawing the diagnostics, verbatim, or "". Kept so
#: an absent summary can name the error that caused it rather than
#: telling the "opened from disk" story regardless.
self._diagnostics_error = ""
#: Whether the table came directly from a run in the current process.
#: This distinguishes disk-loaded results from a live run that failed
#: to return a fitted model.
self._from_live_run = False
#: The :class:`spacr.regression_qc.RegressionQCContext` the residual
#: tabs were drawn from, kept so the homogeneity verdict is read off
#: the SAME arrays the saved report used rather than rebuilt.
self._qc_context = None
#: regression_qc's own scale-location statistics, verbatim.
self._homogeneity = {}
#: The constant-spread verdict now on screen.
self._homogeneity_text = self.NO_HOMOGENEITY_VERDICT
self.clear_diagnostics()
for plot in (self.effect_rank, self.effect_distribution):
plot.set_status(self.NO_EFFECTS_YET)
self.volcano.offer_refit(self.ask_refit)
self._offer_levels()
self._offer_baselines()
self._offer_compartments()
self._colour_by_label.setToolTip("\n\n".join([
self._colour_by_label.toolTip(),
self._colour_by.toolTip(),
self._colour_by_2.toolTip(),
self._colour_by_3.toolTip(),
]))
for field in (self._colour_by, self._colour_by_2, self._colour_by_3):
field.setToolTip("")
from ..screens.settings_model import retarget_field_tooltips
retarget_field_tooltips(self)
[docs]
def set_run_settings(self, settings) -> None:
"""Remember the settings that produced the table now on screen.
Called by the screen when a run finishes, because the run's own
settings are better than anything read back off disk: the saved copy
under ``settings/`` is overwritten by every later run of the same
screen, so on a second run it describes the wrong one.
:param settings: the run's settings mapping, copied; ``None`` or empty
forgets them.
"""
self._run_settings = dict(settings) if settings else None
self._from_live_run = True
self._offer_levels()
if self._model is None:
self.set_summary(None)
[docs]
def ask_refit(self) -> bool:
"""Offer another model for the same data, and ask for the run.
:returns: True if a re-fit was asked for.
The panel goes no further than emitting. It has no worker, no
console and no Stop button, and a widget that started a background
fit with none of those would be a run the user cannot watch or stop.
"""
from ...refit import refit_settings, settings_of_run
base = self._run_settings or settings_of_run(self._path)
try:
refit_settings(base or {})
except ValueError as error:
self.say(str(error))
return False
from .refit_dialog import ask_refit as ask
answer = ask(base, self)
if answer is None:
return False
settings, notes = answer
if notes:
self.say("Re-fitting: " + "; ".join(notes))
self.refit_requested.emit(settings)
return True
#: Message shown when diagnostics have no fitted model. It distinguishes
#: disk-loaded coefficient tables from a live fit and names the available
#: actions.
NO_MODEL_MESSAGE = (
"Residual diagnostics are computed from the fitted model, and only a "
"run in this session hands one over -- a results table read from disk "
"is one row per guide and carries nothing about the wells. Run the "
"regression here, or re-fit from the volcano's right-click menu, and "
"these fill in. The same panels are already on disk beside the "
"results, under regression_qc/.")
#: What the Scale-location tab's verdict says before a fit reaches it.
#: A blank space under a plot reads as "nothing to report", which is the
#: one thing this panel must never say by accident.
NO_HOMOGENEITY_VERDICT = (
"No constant-spread verdict yet: it is computed from the fitted "
"model, and only a run in this session hands one over.")
#: What to reach for when the spread is not constant. NAMED, because
#: "the standard errors are optimistic" without the remedy is a warning
#: a reader can only act on by guessing. ``cov_type`` is already a spaCR
#: setting -- see :data:`spacr.regression_spec` -- so this is a re-fit,
#: not a feature request.
HC3_FIX = (
"The fix: re-fit with cov_type='HC3', a sandwich "
"(heteroscedasticity-consistent) covariance estimator that is valid "
"when the spread is not constant. It is already a supported spaCR "
"setting for ols, wls, glm, poisson, quasi_binomial, logit and "
"probit, and it changes the standard errors and every p-value while "
"leaving the coefficients exactly where they are. Right-click the "
"volcano and choose “Re-fit with another model…”.")
#: What :func:`spacr.regression_qc._panel_scale_location` found, in a
#: sentence. KEYED ON THAT MODULE'S OWN VERDICT STRINGS so the live tab
#: and the saved PDF cannot describe one fit two ways: the statistics come
#: from :func:`spacr.regression_qc.draw_panel` and only the wording is
#: this panel's.
HOMOGENEITY_FINDINGS = {
"no detectable trend in spread":
"CONSTANT SPREAD. The residual spread does not change across the "
"fitted range.",
"variance grows with the fit":
"SPREAD GROWS WITH THE FITTED VALUE.",
"variance shrinks with the fit":
"SPREAD SHRINKS WITH THE FITTED VALUE.",
"spread differs across the fit, but not monotonically":
"SPREAD IS NOT CONSTANT across the fitted range -- a funnel "
"rather than a slope, which a rank correlation on its own "
"reports as nothing at all.",
}
#: What the finding DOES TO THE TABLE, which is the half that changes
#: what a reader does. Only reached when the fit's standard errors are the
#: classical ones; a fit that already used a sandwich estimator gets
#: :data:`ALREADY_ROBUST` instead, because telling its reader their errors
#: are optimistic would be false.
HOMOGENEITY_CONSEQUENCES = {
"no detectable trend in spread":
"The ordinary standard errors in the Summary tab -- and so every "
"p-value in the coefficient table and on the volcano -- are the "
"right ones.",
"variance grows with the fit":
"The standard errors above are OPTIMISTIC, so every p-value in "
"the coefficient table is smaller than it should be and the hits "
"at the top of the volcano gain most from the error.",
"variance shrinks with the fit":
"The standard errors above are wrong in both directions -- "
"conservative where the fit is small, optimistic where it is "
"large -- so the p-values are not comparable across the range.",
"spread differs across the fit, but not monotonically":
"The standard errors above are wrong by an amount that depends "
"on the fitted value, so the p-values are not comparable across "
"the range.",
}
#: The consequence for a fit that was ALREADY given a sandwich estimator.
#: The picture looks the same and means something else: the spread really
#: does vary, and the errors already account for it.
ALREADY_ROBUST = (
"This fit was given cov_type={used!r}, so the standard errors in the "
"Summary tab and every p-value in the table are ALREADY robust to "
"this. The plot is describing the data, not warning about the table.")
[docs]
def diagnostic_plots(self) -> tuple:
"""The three well-level tabs, in the order they are shown.
Public because it is how a caller asks the panel to restyle, export or
interrogate them WITHOUT reaching past the panel into its widgets --
which is how one screen ended up depending on another's private
surface.
"""
return (self.residuals, self.scale_location, self.influence)
#: Live-tile key -> the attribute holding the tab page that shows it.
#: THE GRID'S KEYS, not the tab labels: the grid photographs panels by
#: key and a label is a display string somebody will translate or
#: shorten. Where the page is a splitter carrying the plot plus a table
#: (`Volcano`, `Guide support`) or a plot plus its spread
#: (`Scale-location`), the SPLITTER is named -- `setCurrentWidget` only
#: accepts a widget the tab bar itself owns, so naming the inner plot
#: would be a silent no-op, which is the whole failure this map exists
#: to end.
_PANEL_TABS = {
"regression": "_volcano_tab",
"effect_rank": "effect_rank",
"effect_distribution": "effect_distribution",
"p_values": "p_values",
"qq": "qq",
"controls": "controls",
"agreement": "_support_tab",
"residuals": "residuals",
"scale_location": "_scale_location_tab",
"influence": "influence",
"gene": "gene",
"annotation_check": "annotation_umap",
}
[docs]
def show_panel(self, key: str) -> bool:
"""Raise the tab that holds the live panel named ``key``.
:param key: a key from :meth:`FigureGridView.set_live_tiles`.
:returns: ``True`` when the corresponding tab exists and was raised;
otherwise ``False``. The available tabs depend on the fitted
model and whether the volcano is displayed outside this panel.
"""
attribute = self._PANEL_TABS.get(str(key))
if attribute is None:
return False
page = getattr(self, attribute, None)
if page is None:
return False
try:
if self.tabs.indexOf(page) < 0:
return False
self.tabs.setCurrentWidget(page)
except (RuntimeError, TypeError):
return False
return True
[docs]
def clear_diagnostics(self, reason: str = "") -> None:
"""Empty the three well-level tabs and SAY why they are empty.
:param reason: the specific reason, when there is one. The default is
:data:`NO_MODEL_MESSAGE` -- "no run in this session has fitted
anything yet", which is the ordinary case and still an answer.
THIS DROPS THE MODEL, and that is what it is for: it is called when a
NEW TABLE arrives, and the previous fit has nothing to say about it.
A failure to DRAW the diagnostics is a different event and goes
through :meth:`_clear_diagnostic_views`, which leaves the fit alone --
a failure in the view must not destroy the thing being viewed.
"""
self._model = None
self._diagnostics_error = ""
self._clear_diagnostic_views(reason)
def _clear_diagnostic_views(self, reason: str = "") -> None:
"""Empty the three tabs and say why, WITHOUT discarding the fit."""
self._qc_context = None
for plot in self.diagnostic_plots():
plot._reset_scene()
plot.set_status(reason or self.NO_MODEL_MESSAGE)
self._homogeneity = {}
self._homogeneity_text = (
f"No constant-spread verdict: {reason}" if reason
else self.NO_HOMOGENEITY_VERDICT)
self.homogeneity.setText(self._homogeneity_text)
[docs]
def set_summary(self, model, regression_type=None) -> bool:
"""Fill the Summary tab from the fitted model.
:param model: fitted model whose ``summary()`` is shown, or ``None``
to use the panel's own model, falling back to a summary saved
beside the loaded results.
:returns: True when a summary was rendered.
Rendered from `model.summary()` verbatim rather than rebuilt: the
point of asking for the statsmodels summary is to get the statsmodels
summary, and a re-implementation would differ from every textbook and
every other tool the reader compares it against.
"""
if model is None:
model = self._model
text = summary_text(model, regression_type, path=self._path,
reason=self._no_model_reason())
self._summary.setPlainText(text)
self._name_the_model(model, regression_type)
return not text.startswith("No summary")
def _name_the_model(self, model, regression_type=None) -> str:
"""Update the model identity shown below the volcano plot.
The label and run summary share
(:func:`spacr.regression_summary.model_identity_line`), so the caption
and Summary tab describe the same fitted model.
Returns
-------
str
Rendered identity text, or an empty string when unavailable.
"""
label = getattr(self, "_model_line", None)
if label is None:
return ""
try:
from ...regression_summary import model_identity_line
said = model_identity_line(regression_type,
self._run_settings or {}, model)
except Exception: # noqa: BLE001
LOG.debug("could not name the model", exc_info=True)
said = ""
label.setText(said)
label.setVisible(bool(said))
return said
#: Reason reported when a live run returns no fitted model.
NO_MODEL_FROM_THIS_RUN = (
"this run came back without a fitted model, so there is none to "
"summarise")
def _no_model_reason(self) -> str:
"""Return a verified reason that this panel has no fitted model.
Return an empty string when :func:`summary_text` should infer the
reason from the available path instead.
"""
if self._model is not None:
return ""
if self._diagnostics_error:
return (f"the diagnostics failed and the fit was not kept "
f"({self._diagnostics_error})")
if self._from_live_run:
return self.NO_MODEL_FROM_THIS_RUN
return ""
[docs]
def set_diagnostics(self, model, regression_type=None) -> bool:
"""Fill the residual tabs from the model a run just fitted.
:param model: the fitted model, straight off ``perform_regression``'s
return payload. ``None`` clears the tabs and says why.
:returns: True when the tabs were filled.
EVERY NUMBER COMES FROM :mod:`spacr.regression_qc`. The residuals, the
standardisation, the leverage and Cook's distance are the arrays that
module already computes for the report it writes to disk -- so the
live tab and the saved PDF cannot disagree about which well is
influential, which they would within a week if this panel did its own
arithmetic.
NEVER RAISES INTO THE GUI. A model class that cannot be diagnosed --
the penalised backends keep no design matrix -- puts its reason on the
three tabs instead, because "this fit cannot answer that" and "this
tab is broken" must not look the same.
"""
if model is None:
self.clear_diagnostics()
return False
self._model = model
self._diagnostics_error = ""
try:
from ...regression_qc import (PanelUnavailable, context_from_model,
cooks_distance)
except Exception as error: # noqa: BLE001
self._diagnostics_error = f"could not load the diagnostics module: {error}"
self._clear_diagnostic_views(
f"Could not load the diagnostics module: {error}")
return False
try:
ctx = context_from_model(model, coef_df=self._frame,
regression_type=regression_type)
except PanelUnavailable as error:
self._diagnostics_error = str(error)
self._clear_diagnostic_views(str(error))
return False
except Exception as error: # noqa: BLE001
self._diagnostics_error = (
f"{type(error).__name__}: {error}")
self._clear_diagnostic_views(
f"The diagnostics could not be computed for this fit "
f"({type(error).__name__}: {error}).")
return False
self._qc_context = ctx
labels = list(ctx.labels) if ctx.labels is not None else []
reason = (ctx.standardisation.reason
if ctx.standardisation is not None else "")
self.residuals.set_residuals(ctx.fitted, ctx.resid, labels=labels)
self.scale_location.set_scale_location(
ctx.fitted, ctx.std_resid, labels=labels, reason=reason or "")
self.influence.set_influence(
ctx.leverage, ctx.std_resid,
cooks_distance(ctx.std_resid, ctx.leverage, ctx.p),
labels=labels, n_params=ctx.p, reason=reason or "")
self.judge_homogeneity(ctx)
self._note_the_diagnostics()
return True
[docs]
def judge_homogeneity(self, ctx=None) -> str:
"""Say whether the residual spread is constant, and what to do if not.
:param ctx: a :class:`spacr.regression_qc.RegressionQCContext`.
``None`` re-judges the context the diagnostics were last drawn
from, so a caller does not have to keep a copy of it to ask again.
:returns: the verdict now on screen.
NOT ONE NUMBER RE-DERIVED HERE. The statistics come from
:func:`spacr.regression_qc.draw_panel` -- Spearman's rho on
sqrt|standardised residual| against fitted, a Brown-Forsythe test
across quartiles of the fitted value, and the quartile SD ratio --
drawn into a throwaway axes purely to get the dict it returns. That
module is 3,444 lines and already computes them for the PDF the run
writes; a second implementation here would disagree with it about the
same fit inside a week, and the reader would have no way to tell which
of the two was wrong.
TWO STATISTICS, NOT ONE, and that is regression_qc's decision rather
than this panel's: a rank correlation is exactly zero for a SYMMETRIC
funnel -- wide at both ends, narrow in the middle -- which is what a
mis-specified link produces, and a panel reporting "no trend in
spread" over one would be a confident wrong answer.
"""
from matplotlib.figure import Figure
from ...regression_qc import PanelUnavailable, draw_panel
ctx = self._qc_context if ctx is None else ctx
if ctx is None:
self._homogeneity = {}
self._homogeneity_text = self.NO_HOMOGENEITY_VERDICT
self.homogeneity.setText(self._homogeneity_text)
return self._homogeneity_text
figure = Figure()
try:
stats = draw_panel("scale_location", ctx, figure.add_subplot(111))
except PanelUnavailable as error:
self._homogeneity = {}
self._homogeneity_text = f"No constant-spread verdict: {error}"
except Exception as error: # noqa: BLE001
self._homogeneity = {}
self._homogeneity_text = (
f"No constant-spread verdict: the test could not be computed "
f"for this fit ({type(error).__name__}: {error}).")
else:
self._homogeneity = dict(stats)
self._homogeneity_text = self._homogeneity_sentence(stats)
finally:
figure.clf()
self.homogeneity.setText(self._homogeneity_text)
index = self.tabs.indexOf(self._scale_location_tab)
if index >= 0:
self.tabs.setTabToolTip(index, self._homogeneity_text)
return self._homogeneity_text
[docs]
def homogeneity_verdict(self) -> str:
"""Whether the residual spread is constant, in the words on screen.
Public because it is the one sentence on the Scale-location tab that
changes what a reader does, and a test that reads it back has to be
able to do so without reaching into a label.
"""
return self._homogeneity_text
[docs]
def homogeneity_stats(self) -> dict:
"""The numbers behind the verdict -- ``regression_qc``'s own dict.
``spearman_rho``, ``spearman_p``, ``levene_p``,
``quartile_sd_ratio``, ``verdict`` and ``n_points``, exactly as
:func:`spacr.regression_qc.draw_panel` returned them for the saved
scale-location panel. Empty when there is no verdict.
"""
return dict(self._homogeneity)
def _homogeneity_sentence(self, stats) -> str:
"""regression_qc's finding, what it costs, the fix, and the numbers."""
verdict = str(stats.get("verdict", ""))
finding = self.HOMOGENEITY_FINDINGS.get(
verdict,
f"The constant-spread test returned “{verdict}”, which this "
f"panel has no sentence for; the numbers are below.")
numbers = (
f"Spearman rho = {stats.get('spearman_rho', float('nan')):+.2f} "
f"(p = {stats.get('spearman_p', float('nan')):.2g}), "
f"Brown-Forsythe p = {stats.get('levene_p', float('nan')):.2g}, "
f"max/min quartile SD = "
f"{stats.get('quartile_sd_ratio', float('nan')):.2f}, over "
f"{stats.get('n_points', 0)} wells. These are the same numbers "
f"the saved regression_qc/scale_location panel prints.")
return " ".join(part for part in
(finding, self._consequence(verdict), numbers) if part)
def _consequence(self, verdict) -> str:
"""What the finding costs the table, and the fix when there is one.
THE FIT'S OWN COVARIANCE IS READ FIRST. A model already given
``cov_type='HC3'`` has robust standard errors, so "the errors above
are optimistic" would be simply false about it and "re-fit with HC3"
would be advice to repeat what was already done. Read off the model
rather than off the settings, because the settings copy on disk is
overwritten by every later run of the same screen.
"""
used = str(getattr(self._model, "cov_type", "") or "")
robust = bool(used) and used.lower() not in ("nonrobust", "none")
flat = verdict == "no detectable trend in spread"
if flat or not robust:
consequence = self.HOMOGENEITY_CONSEQUENCES.get(verdict, "")
else:
consequence = self.ALREADY_ROBUST.format(used=used)
if flat or robust:
return consequence
return f"{consequence} {self.HC3_FIX}".strip()
def _keyed_plots(self) -> tuple:
"""Every plot whose marks are individual coefficients or genes.
The volcano, the effect ranking, the Q-Q, the control panel and the
guide-agreement plot. The two HISTOGRAMS are deliberately not here:
their marks are bins of many rows, so neither can select one and
neither pretends to. The residual plot is not here either -- its
points are wells, not coefficients, and there is no key for a well.
"""
return (self.volcano, self.effect_rank, self.qq, self.controls,
self.agreement)
def _show_keys(self, keys) -> None:
"""A set of coefficients was chosen on a plot: narrow the table."""
self.table.show_keys(list(keys))
def _extra_colour_column(self, combo, first):
"""The column on a secondary channel, or ``None``.
THREE WAYS IT IS NONE, and each is a real answer rather than an
oversight:
* nothing is chosen -- the ordinary case, one channel;
* the FIRST channel is not in force, because a shape that means one
thing beside a colour that means the q-value is two claims on one
dot with nothing on screen saying which;
* this channel repeats a column already encoded, which draws the
same fact twice and tells the reader nothing the first channel
did not.
"""
if not first:
return None
chosen = combo.currentData() if combo is not None else None
if not chosen or chosen == first:
return None
if combo is self._colour_by_3 and \
chosen == self._colour_by_2.currentData():
return None
if chosen == self.LOPIT_KEY:
return None
return str(chosen)
[docs]
def colour_channels(self) -> list:
"""The columns currently encoded, hue first. For tests and state."""
first = self._colour_by.currentData()
out = [first] if first else []
for combo in (self._colour_by_2, self._colour_by_3):
column = self._extra_colour_column(combo, first)
if column:
out.append(column)
return out
def _select_many_from_a_plot(self, keys) -> None:
"""A band or a modifier-click chose several: select them all.
SELECT, NOT NARROW -- see :meth:`_show_keys` for the other one. A
band is a comparison, and hiding everything the user did not enclose
would remove the thing they are comparing against.
"""
keys = [str(k) for k in (keys or ())]
if len(keys) < 2:
return
self._selected_key = keys[-1]
self.table.select_keys(keys)
[docs]
def selected_keys(self) -> list:
"""Return identifiers selected in the results table.
If the table has no active multiselection, return the most recent
single selection used by linked result views.
"""
keys = list(self.table.selected_keys())
if keys:
return keys
return [self._selected_key] if self._selected_key else []
[docs]
def results_frame(self):
"""Return the complete coefficient table for the displayed run.
The table is unaffected by the gene/guide view filter. Use
:meth:`filtered_frame` for the rows currently shown by the result tabs.
"""
return self._frame
[docs]
def run_folder(self) -> str:
"""Return the folder for the regression run shown by this panel.
Live results may store a directory while results loaded from disk may
store a table path; both resolve to the containing run folder. Return
an empty string for an unassociated CSV or an in-memory frame.
"""
return self._folder_of(self._path)
@staticmethod
def _folder_of(source) -> str:
"""The run folder behind ``source``, whatever shape it arrived in.
A run opened off disk gives the CSV; a live run gives the directory
``perform_regression`` wrote. ONE ANSWER FOR BOTH, because they are
the same run -- and while they were two answers they were also two
keys, so a run looked at live and then returned to from the Runs tab
was two runs to everything keyed on this.
"""
path = str(source or "").strip()
if not path:
return ""
path = os.path.abspath(os.path.expanduser(path))
if os.path.isdir(path):
return path
if os.path.isfile(path):
return os.path.dirname(path)
if os.path.splitext(os.path.basename(path))[1]:
return os.path.dirname(path)
return path
[docs]
def run_name(self) -> str:
"""The run on screen, as the name the Runs tab calls it.
``results/<kind>_<n>`` is the folder a run writes, so the basename is
``ols_3`` -- which is what the Runs table shows, what the figure grid
heads its section with and what the montage names. One vocabulary
prevents three views from calling the same run three different things.
"""
folder = self.run_folder()
return os.path.basename(folder.rstrip(os.sep)) if folder else ""
def _name_the_run(self) -> None:
"""Put the run's name in the header, beside its table.
LABELLED, not bare. `ols_4` on its own between a status sentence and
a "colour by" menu is a word with no job; "Run: ols_4" is the answer
to the question the user is actually asking of this header.
"""
name = self.run_name()
self._run_label.setText(f"Run: {name}" if name else self.NO_RUN_NAMED)
self._run_label.setToolTip(
self.run_folder() if name else
"The table on screen was not read from a run folder.")
[docs]
def status_text(self) -> str:
"""Whatever the panel last had to say, success or failure.
The header label is the only place the user learns why a table is
empty, so it is worth reading back in a test rather than reaching into
a private widget.
"""
return self._status
[docs]
def say(self, text: str, detail: str = "") -> None:
"""Put a sentence where the user will read it.
Public because the panel is not the only thing that can fail to fill
it: the caller that decides WHICH folder to hand over can come up with
nothing at all, and that has to reach the same header rather than
being logged at debug level and dropped.
:param text: concise status shown in the results header and retained
for :meth:`status_text`.
:param detail: the tooltip; the long form, when there is one.
"""
self._status = text
self._source.setText(text)
self._source.setToolTip(detail or text)
[docs]
def browse_for_results(self) -> bool:
"""Ask for a results folder and load it. False if nothing was chosen.
Companion to :meth:`load`: the panel is otherwise only ever filled by
a run finishing, which is no use to a user who has results and no run.
While a load is already in flight the same button is the CANCEL, so
the one control that starts a read is also the one that abandons it;
a read with no way out is the freeze this loader was moved off the
GUI thread to remove.
"""
from PySide6.QtWidgets import QFileDialog
if self._loading:
return self.cancel_load()
start = ""
if self._path:
start = os.path.dirname(str(self._path))
folder = QFileDialog.getExistingDirectory(
self, "Choose a regression results folder", start)
if not folder:
return False
return self.load(folder)
[docs]
def closeEvent(self, event): # noqa: N802 - Qt's spelling
"""Stop the results loader before closing the widget.
Qt requires a running ``QThread`` to outlive every object that owns it.
The bounded shutdown cancels the load and waits briefly; a slower
worker is transferred to :func:`spacr.qt.bridge.drain_thread` so that
closing the screen neither terminates the worker nor blocks on it.
:param event: Qt close event passed to the parent implementation.
"""
try:
self._load_jobs.shutdown()
except Exception: # noqa: BLE001
LOG.debug("could not stop the results loader", exc_info=True)
super().closeEvent(event)
[docs]
def is_loading(self) -> bool:
"""Whether a run is being read right now."""
return bool(self._loading)
def _set_loading(self, loading: bool) -> None:
"""Record whether a read is in flight, and offer the way out of it.
The one button that starts a read becomes the cancel while the read
is running, so a slow folder can always be abandoned.
:param loading: whether a run is being read right now.
"""
self._loading = bool(loading)
self._load_generation = getattr(self, "_load_generation", 0) + 1
button = getattr(self, "_load_button", None)
if button is None:
return
try:
if self._loading:
button.setText("Cancel load")
button.setToolTip(
"Stop reading this run. The run already on screen stays "
"on screen.")
else:
button.setText("Load results\u2026")
button.setToolTip(
"Choose a regression results folder, or a parent of one. "
"The most recently written results table under it is "
"loaded.")
except RuntimeError: # noqa: BLE001
pass
[docs]
def cancel_load(self) -> bool:
"""Cancel an active result load while preserving the current view.
:returns: Whether a load was active.
"""
if not self._loading:
return False
try:
self._load_jobs.cancel()
except Exception: # noqa: BLE001
LOG.debug("could not cancel the results loader", exc_info=True)
self._set_loading(False)
self.say("Loading was cancelled; the run already on screen is "
"unchanged.")
self.load_finished.emit(False)
return True
[docs]
def start_load(self, path) -> bool:
"""Start loading a regression run outside the GUI thread.
File discovery and CSV reads run in a worker. Only the resulting data
is returned to the GUI thread.
:param path: Regression-results directory to load.
:returns: whether a load was started. ``False`` when one already is --
a second click must not read the same folder twice.
"""
if self._loading:
return False
if not path:
self.say("Nothing was handed to the results panel, so there is "
"no folder to search. Use “Load results…”.")
return False
self._set_loading(True)
generation = self._load_generation
self._on_load_progress(generation, 1, 0, 0,
os.path.basename(str(path)) or str(path))
started = self._load_jobs.submit(
lambda: self._read_run(
path, progress=lambda step, done, total, name:
self._relay_load_progress(generation, step, done, total,
name)),
self._finish_load)
if not started:
self._set_loading(False)
return bool(started)
def _relay_load_progress(self, generation, step, done, total,
name) -> None:
"""Called BY THE WORKER. Emits one stage of the load, and nothing else.
Guarded the way `JobRunner._relay` is: a panel closed while a read is
still running takes its C++ half with it, and the emit then raises
``RuntimeError`` inside the worker.
"""
try:
self._load_progress_relayed.emit(int(generation), int(step),
int(done), int(total), str(name))
except RuntimeError:
pass
def _load_stage_text(self, step, done=0, total=0, name="") -> str:
"""The sentence for one stage of a load, counted against the stages.
:param step: 1 searching, 2 reading, 3 building the views.
:param done: for the reading step, the file being read; else 0.
:param total: for the reading step, the files beside the run's
primary table, counted with it; else 0.
:param name: the folder searched, the file read, or the row count.
"""
steps = self.LOAD_STEPS
if step == 1:
return tr("Step {step} of {steps}: searching {folder} for a "
"results table…", step=1, steps=steps, folder=name)
if step == 2:
return tr("Step {step} of {steps}: reading {name} (file {done} "
"of {total})…", step=2, steps=steps, name=name,
done=done, total=total)
return tr("Step {step} of {steps}: building the table, plots and "
"diagnostics for {rows} rows…", step=3, steps=steps,
rows=name)
def _on_load_progress(self, generation, step, done, total,
name) -> None:
"""Say which stage of the load is running. Always on the GUI thread.
A stage from a load that has since finished, failed or been cancelled
is dropped: the generation it carries is no longer the current one,
and a late "reading results.csv" must not overwrite the sentence that
says the load was cancelled.
"""
if generation != self._load_generation or not self._loading:
return
self.say(self._load_stage_text(step, done, total, name))
def _say_building(self, frame) -> None:
"""Announce step 3 and paint it before the GUI thread is busy.
Building the views runs on the GUI thread, so the sentence is painted
at once rather than queued behind the work it announces.
"""
self.say(self._load_stage_text(3, name=f"{len(frame):,}"))
try:
self._source.repaint()
except RuntimeError:
pass
@staticmethod
def _read_run(path, progress=None):
"""THE WORKER HALF. Touches no widget; returns what the GUI needs.
Returns a dict rather than raising, because a JobRunner job that
raises loses the detail on the way back across the thread boundary --
the same reason `_merge_worker` returns its outcome.
:param path: the folder or table to load.
:param progress: optional ``progress(step, done, total, name)``,
called as the read reaches each file; see
:meth:`_load_stage_text`.
"""
import pandas as pd
searched = os.path.abspath(os.path.expanduser(os.fspath(path)))
if not os.path.exists(searched):
return {"error": f"{searched} does not exist, so there is "
f"nothing to load from it."}
tables = find_results_tables(searched)
if not tables:
return {"error": (
f"Searched {searched} and found none of "
f"{', '.join(RESULT_FILENAMES)} in it or in any folder up to "
f"{MAX_SEARCH_DEPTH} deep.")}
reading = None
if progress is not None:
def reading(done, total, name):
"""Pass one file of the read on as step 2."""
progress(2, done, total, name)
try:
frame, found, merged = read_run_tables(tables, progress=reading)
except Exception as error: # noqa: BLE001 - report, do not raise
return {"error": f"Could not read {tables[0]}: {error}"}
return {"frame": frame, "found": found, "searched": searched,
"tables": tables, "merged": merged}
def _finish_load(self, outcome) -> bool:
"""THE GUI HALF: everything that touches a widget."""
self._set_loading(False)
ok = False
if not isinstance(outcome, dict):
self.say("The run loader came back with nothing, which is a bug "
"in the loader rather than in the run.")
elif outcome.get("error"):
self.say(str(outcome["error"]))
else:
self._say_building(outcome["frame"])
ok = self._apply_loaded_run(
outcome["frame"], outcome["found"], outcome["searched"],
outcome["tables"])
self.load_finished.emit(bool(ok))
return ok
def _on_load_job_failed(self, message: str) -> None:
"""Report a failed load and announce that loading finished unsuccessfully.
:param message: the failure text from the job runner.
"""
self._set_loading(False)
self.say(f"The run could not be read: {message}")
self.load_finished.emit(False)
[docs]
def load(self, path) -> bool:
"""Load a results CSV, a run folder, or a parent of one.
THE SYNCHRONOUS ENTRY POINT, kept for tests and headless callers.
The GUI goes through :meth:`start_load`; both end in
:meth:`_apply_loaded_run`, so the two cannot drift -- which is the
rule `start_merge` and `merge` already follow.
EVERY WAY THIS FAILS SAYS SO. It used to return False five different
ways and leave the user looking at a table with columns and no rows,
which is indistinguishable from a run that produced nothing.
:param path: results CSV, run folder or a folder above one (searched
up to :data:`MAX_SEARCH_DEPTH` deep); ``~`` is expanded.
"""
import pandas as pd
if not path:
self.say("Nothing was handed to the results panel, so there is "
"no folder to search. Use “Load results…”.")
return False
searched = os.path.abspath(os.path.expanduser(os.fspath(path)))
if not os.path.exists(searched):
self.say(f"{searched} does not exist, so there is nothing to "
f"load from it.")
return False
tables = find_results_tables(searched)
if not tables:
self.say(
f"Searched {searched} and found none of "
f"{', '.join(RESULT_FILENAMES)} in it or in any folder up to "
f"{MAX_SEARCH_DEPTH} deep.")
return False
try:
frame, found, merged = read_run_tables(tables)
except Exception as error: # noqa: BLE001 - report, do not raise
self.say(f"Could not read {tables[0]}: {error}")
return False
return self._apply_loaded_run(frame, found, searched, tables)
def _apply_loaded_run(self, frame, found, searched, tables) -> bool:
"""Put a read run on screen. THE ONE ENDING BOTH LOAD PATHS SHARE.
Split out so `load` (synchronous) and `start_load` (on a worker)
cannot drift -- the rule `merge` and `start_merge` already follow, and
the reason the run loader was still on the GUI thread long after the
merge had been moved off it is that nobody had made them one path.
"""
if not self.set_frame(frame, source=found):
self.say(f"{self._status} ({found}, found under {searched})")
return False
runs = {os.path.dirname(table) for table in tables}
if len(runs) > 1:
self.say(
f"{self._status} — newest of {len(runs)} runs under "
f"{searched}",
detail="\n".join(tables[:20]))
elif len(tables) > 1:
self.say(f"{self._status} — the newest run under {searched} "
f"wrote {len(tables)} tables; this is the full one.",
detail="\n".join(tables[:20]))
return True
def _say_if_no_p_values(self, frame) -> None:
"""Explain a table whose coefficients carry no significance.
Distinguishes the two ways it happens, because they call for
different things from the reader: a mixed fit's guide rows are BLUPs
and never had a p value, and that is expected; anything else with no
finite p value is a fit that failed to produce one, and that is not.
"""
import pandas as pd
if "p_value" not in frame.columns:
return
values = pd.to_numeric(frame["p_value"], errors="coerce")
if values.notna().any():
return
kinds = set()
if "term_type" in frame.columns:
kinds = {str(k) for k in frame["term_type"].dropna().unique()}
if kinds and kinds <= {"random_effect_blup"}:
self.say(
f"These {len(frame):,} row(s) are BLUPs, not estimates. A "
f"mixed model makes the guide a RANDOM effect, so each guide "
f"gets a shrunken prediction and no p value -- which is why "
f"the volcano is empty, its vertical axis being the p value. "
f"The coefficients are in the table beside it. For per-guide "
f"significance, fit at guide level with a fixed-effect "
f"model, or use inference='nonparametric', which tests each "
f"guide on its own.")
else:
self.say(
f"None of these {len(frame):,} coefficient(s) carries a "
f"p value, so nothing can be plotted against significance. "
f"That is a fit that did not produce one rather than a fit "
f"with nothing to say -- the coefficients themselves are in "
f"the table.")
@staticmethod
def _name_the_levels(frame):
"""``frame`` with a ``level`` column beside ``feature``, if needed.
WHATEVER TELLS A GENE FROM ITS GUIDES HAS TO BE IN THE TABLE, because
the table is what leaves the panel. The report that pinned the level
in the first place was made against a picture and a report, not
against the panel's internal state -- so a distinction the panel knows
and the exported rows do not is a distinction that has not been made.
Current runs write the column themselves and are left alone. Older
tables carry the same fact in the term name, and this materialises it:
``gene_fraction:gene[225160]`` becomes level ``gene`` and
``fraction:grna[225160_1]`` level ``grna``, in a column the
coefficient table shows, "Copy" copies, and the volcano offers on its
colour-by menu so the two families can be told apart at a glance.
Only when the table actually holds BOTH families. A guide-only run
needs no column to distinguish rows that are all the same kind, and a
constant column is clutter on every tab that shows it.
"""
try:
from ...hits import coefficient_levels
except Exception:
return frame
columns = list(getattr(frame, "columns", ()))
if "level" in columns or "feature" not in columns:
return frame
levels = coefficient_levels(frame)
if not (levels == "gene").any() or not (levels == "grna").any():
return frame
frame = frame.copy()
levels = levels.replace("", "nuisance")
frame.insert(columns.index("feature") + 1, "level", levels)
return frame
[docs]
def set_frame(self, frame, source: str = "") -> bool:
"""Show an already-loaded coefficient table.
:param frame: the coefficient DataFrame; ``None`` or an empty frame is
reported in the status line and returns ``False``.
"""
import pandas as pd
if frame is None or not len(frame):
self.say("The results table is empty: it has columns but no "
"rows, so the fit produced no coefficients.")
return False
self._say_if_no_p_values(frame)
frame = self._name_the_levels(frame)
self._remember_plot_state()
self._frame = frame
self._path = source
self._name_the_run()
self._ranking = self._rank_by(frame, source)
self._from_live_run = False
self.clear_diagnostics()
self.set_summary(None)
from ...refit import settings_of_run
try:
self._run_settings = settings_of_run(source) if source else None
except Exception: # noqa: BLE001
self._run_settings = None
self._colour_by.blockSignals(True)
self._colour_by.clear()
self._colour_by.addItem("nothing", None)
skipped = []
cap = max(40, len(frame) // 20)
for name in frame.columns:
dtype = frame[name].dtype
text_or_category = (
pd.api.types.is_object_dtype(dtype)
or pd.api.types.is_string_dtype(dtype)
or isinstance(dtype, pd.CategoricalDtype)
)
if text_or_category:
distinct = frame[name].nunique(dropna=True)
if 1 < distinct <= cap:
self._colour_by.addItem(f"{name} ({distinct})", name)
elif name in self.ALWAYS_OFFERED:
self._colour_by.addItem(f"{name} ({distinct})", name)
skipped.append(
f"{name} has {distinct} distinct value"
f"{'' if distinct == 1 else 's'}"
+ (" -- the control names matched no feature"
if distinct == 1 and name == "condition" else ""))
try:
from ...localisation import present
compartments = present(frame)
except Exception: # noqa: BLE001
compartments = []
if compartments:
self._colour_by.addItem(
f"LOPIT localisation ({len(compartments)})", self.LOPIT_KEY)
preferred = self._colour_by.findData("condition")
self._colour_by.setCurrentIndex(preferred if preferred >= 0 else 0)
self._colour_by.blockSignals(False)
for extra in (self._colour_by_2, self._colour_by_3):
keep = extra.currentData()
extra.blockSignals(True)
extra.clear()
for index in range(self._colour_by.count()):
extra.addItem(self._colour_by.itemText(index),
self._colour_by.itemData(index))
back = extra.findData(keep) if keep is not None else 0
extra.setCurrentIndex(back if back >= 0 else 0)
extra.blockSignals(False)
self._colour_by_note = "; ".join(skipped)
self._selected_key = None
try:
blocked = self.table.table.blockSignals(True)
self.table.table.clearSelection()
self.table.table.setCurrentCell(-1, -1)
self.table.table.blockSignals(blocked)
except (RuntimeError, AttributeError):
pass
for plot in self._keyed_plots():
plot.clear_highlight()
for histogram in (self.p_values, self.effect_distribution):
histogram.clear_highlight()
self.gene.clear()
self.gene.warm_for(frame)
self._compartment = None
self._threshold_method = DEFAULT_THRESHOLD_METHOD
self._threshold_multiplier = DEFAULT_THRESHOLD_MULTIPLIER
self._p_value_kind = "raw"
try:
self.volcano.auto_range_axes()
except Exception: # noqa: BLE001
LOG.debug("could not release the previous run's axis limits",
exc_info=True)
self._level = self._default_level()
self._offer_levels()
self._offer_p_values()
self._offer_thresholds()
self._offer_compartments()
self.refresh_views()
self._mark_the_level_on_the_plots()
note = self.both_levels_note()
if note:
self.say(f"{self._status} {note}")
if self._colour_by_note:
self.say(f"{self._status} Colouring: {self._colour_by_note}.")
self._restore_plot_state(source)
self.loaded.emit(source or "")
return True
[docs]
def plot_state(self) -> dict:
"""What the user has built on the plot, as data.
The design: "every regression run should have its own
interactive volcano plot." A run does not have a volcano today -- the
screen has one and a run borrows it -- so opening run B destroyed the
level, the colouring, the axis limits, the effect cut and the
selection a user had chosen on run A.
EVERYTHING HERE BELONGS TO THE RUN AND NOT TO THE WIDGET. The colour
column is stored by NAME rather than by index, because the combo box
is rebuilt from each table's own columns and index 3 is a different
column in the next run.
"""
reader = getattr(self.volcano, "pinned_limits", None)
pinned = reader() if callable(reader) else (
getattr(self.volcano, "_pinned", None) or {})
return {
"level": self._level,
"colour_by": self._colour_by.currentData(),
"baseline": tuple(self._baseline),
"compartment": self._compartment,
"p_value_kind": self._p_value_kind,
"threshold_method": self._threshold_method,
"threshold_multiplier": self._threshold_multiplier,
"selected_key": self._selected_key,
"x_limits": pinned.get("x"),
"y_limits": pinned.get("y"),
}
[docs]
def apply_plot_state(self, state) -> bool:
"""Put a saved :meth:`plot_state` back. Returns whether it applied.
ONE REDRAW, not one per setting. Every public setter here redraws,
so restoring nine of them through nine setters would draw the panel
nine times on every run switch -- and the intermediate frames are of
combinations the user never chose.
:param state: dict as :meth:`plot_state` returns it; keys it lacks
keep their current value. Nothing applies when it is not a dict or
no table is loaded.
"""
if not isinstance(state, dict) or self._frame is None:
return False
if "level" in state:
self._level = state["level"]
if "baseline" in state:
kind, name = tuple(state["baseline"])
self._baseline = (kind, name)
for field in ("compartment", "p_value_kind", "threshold_method",
"threshold_multiplier"):
if field in state:
setattr(self, f"_{field}", state[field])
colour = state.get("colour_by")
if colour is not None:
index = self._colour_by.findData(colour)
if index >= 0:
blocked = self._colour_by.blockSignals(True)
self._colour_by.setCurrentIndex(index)
self._colour_by.blockSignals(blocked)
self._offer_levels()
self._offer_p_values()
self._offer_thresholds()
self._offer_compartments()
self._offer_baselines()
self.refresh_views()
self._mark_the_level_on_the_plots()
limits = (state.get("x_limits"), state.get("y_limits"))
if any(limit is not None for limit in limits):
try:
self.volcano.set_axis_limits(x=limits[0], y=limits[1])
except Exception: # noqa: BLE001
LOG.debug("could not restore the run's axis limits",
exc_info=True)
key = state.get("selected_key")
if key:
try:
self._select_key(str(key))
except Exception: # noqa: BLE001
LOG.debug("could not restore the run's selection",
exc_info=True)
return True
[docs]
def remembered_runs(self) -> tuple:
"""Return run paths whose plot state has been stored.
The currently displayed run is omitted because its state remains
live in the widgets until the panel switches away from it.
"""
return tuple(self._plot_states)
[docs]
def workspace_state(self) -> dict:
"""Return saved view state for every run opened in this panel.
The currently displayed run is recorded before the stored run states
are copied into the returned workspace mapping.
"""
self._remember_plot_state()
return {
"path": str(self._path or ""),
"level": self._level,
"runs": {key: dict(state) for key, state in self._plot_states.items()},
}
[docs]
def apply_workspace_state(self, state) -> bool:
"""Put the remembered views back, and reopen the run that was open.
Returns whether anything was put back. THE STORE IS MERGED, NOT
REPLACED: restoring a workspace into a session that already has runs
open must not silently drop the views the user built since. A run in
both is taken from the document, which is what the user asked to
restore.
:param state: dict as :meth:`workspace_state` returns it; its
``"runs"`` plot states are merged in and its ``"path"`` is loaded
again when it still exists. A non-dict applies nothing.
"""
if not isinstance(state, dict):
return False
runs = state.get("runs")
if isinstance(runs, dict):
for key, remembered in runs.items():
if isinstance(remembered, dict):
self._plot_states[str(key)] = dict(remembered)
path = str(state.get("path") or "")
if path and os.path.exists(path):
return bool(self.load(path))
return bool(runs)
[docs]
def forget_plot_state(self, source) -> bool:
"""Drop one run's remembered plot. The design deletes a run.
:param source: the run's folder, or any path inside it -- the CSV a
caller happens to be holding answers the same as the folder.
:returns: whether there was anything to drop.
A deleted run must take its state with it, or a later run written
into the same folder inherits the deleted one's level and colouring
and there is nothing on screen saying where they came from.
"""
return self._plot_states.pop(self._plot_state_key(source),
None) is not None
[docs]
def forget_run(self, source) -> bool:
"""A run was deleted: drop its view, and clear the panel if it is IT.
THE ORDER IS THE WHOLE POINT, and the failure sequence is explicit
before 146 existed: "deleting the run CURRENTLY ON SCREEN must also
clear the panel, or leaving that run re-saves the state that was just
forgotten." `set_frame` calls `_remember_plot_state` on the way out of
a run, so forgetting first and clearing second files the deleted run
again, under the same key, with the state it was just relieved of.
So the panel lets go of the run FIRST -- `_path` and `_frame` cleared
by hand rather than through `set_frame`, which is a route INTO a run
and re-saves on the way -- and forgets afterwards.
:param source: the run's folder, or any path inside it.
:returns: whether anything was dropped: a state, the table, or both.
"""
key = self._plot_state_key(source)
was_showing = bool(key) and self._plot_state_key(self._path) == key
if was_showing:
self._path = ""
self._frame = None
self.clear_diagnostics(
"the run this was describing has been deleted")
self.say("The run this was showing has been deleted. "
"Pick another in the Runs tab.")
return bool(self.forget_plot_state(source)) or was_showing
@classmethod
def _plot_state_key(cls, source) -> str:
"""Return the key used to store a results table's plot state.
A live run path and its canonical ``results.csv`` file share the run
folder as their key. Alternative tables in the same folder include
their file name, which keeps gene- and guide-level plot selections
independent. Sources without a resolvable folder use their string
representation.
"""
folder = cls._folder_of(source)
if not folder:
return str(source or "")
name = os.path.basename(str(source or ""))
if name and name != os.path.basename(folder) and name != _CANONICAL_TABLE:
return f"{folder}::{name}"
return folder
def _remember_plot_state(self) -> None:
"""Save the run on screen, if it has a path to be saved under."""
key = self._plot_state_key(self._path)
if key and self._frame is not None:
self._plot_states[key] = self.plot_state()
def _restore_plot_state(self, source: str) -> bool:
"""Put back what this run was left looking like, if anything."""
state = self._plot_states.get(self._plot_state_key(source))
return self.apply_plot_state(state) if state else False
def _mark_the_level_on_the_plots(self) -> None:
"""Put the level sentence where the dots are, not only in a header.
A status line at the top of a panel is read once, on load. The
question "am I looking at guides or genes" is asked every time the
user comes back to the tab, and the answer belongs beside the marks
it describes.
"""
self._offer_levels()
[docs]
def refresh_views(self) -> None:
"""Draw EVERY tab from the coefficient table at the chosen level.
ONE PIECE OF STATE, READ SIX TIMES. `_level` used to reach the volcano
and nothing else, so "genes only" left the coefficient table, the
p-value histogram, the Q-Q, the control panel and the guide support
showing the whole fit -- five tabs disagreeing with the sixth, at the
same time, with nothing on screen saying which was which. That is
worse than no filter at all: a reader who trusts the volcano and reads
the inflation figure off the Q-Q beside it has combined two different
multiple-testing families and cannot tell.
Public because it is also what a caller does after changing something
the panel does not own -- and because a private redraw that four
methods have to remember to call is how the fifth one forgets.
"""
frame = self._frame
if frame is None:
self._say_which_family()
return
shown = self.filtered_frame()
kind, column = self._ranking
self._redraw_volcano()
self.table.configure(significance_filter=(kind == "p-value"))
significance = None
if kind == "p-value" and not any(
name in shown.columns
for name in ("q_value", "adjusted_p_value", "p_value")):
significance = column
self.table.set_frame(
for_table(shown), key_column=self._key_column(shown),
significance_column=significance)
self._show_significance(shown, kind, column, self._path or "")
self._draw_effects(shown, kind)
self._draw_controls(shown)
self._draw_guide_support(frame)
self._say_which_family()
def _analysis_path(self) -> str:
"""``'permutation'`` when this run permuted, else ``'fitted'``.
Read off the TABLE, not the settings: a folder opened from disk may
carry no settings at all, and the columns a permutation writes are
proof it ran.
"""
columns = list(getattr(self._frame, "columns", ()))
if any(name in columns for name in PERMUTATION_COLUMNS):
return "permutation"
if columns:
return "fitted"
settings = self._run_settings if isinstance(self._run_settings,
dict) else {}
mode = str(settings.get("analysis_mode") or "").strip().lower()
if mode == "guide_permutation":
return "permutation"
return "fitted"
@staticmethod
def _gene_terms(frame) -> dict:
"""``{gene id: the gene-level term that names it}``.
The bridge between the two things a screen calls a gene. The support
table is indexed by the bare id (``244480``); the coefficient table,
the volcano and the gene tile all join on the design-matrix term
(``gene_fraction:gene[244480]``). Without this the agreement plot
would be a second key space that nothing else can resolve, and
clicking a gene there would select nothing anywhere.
"""
if frame is None or "feature" not in getattr(frame, "columns", ()):
return {}
try:
from ...hits import gene_of
except Exception:
return {}
terms = {}
for feature in frame["feature"].astype(str):
if not feature.startswith("gene_fraction:gene"):
continue
gene = gene_of(feature)
if gene is not None:
terms.setdefault(str(gene), feature)
return terms
def _draw_guide_support(self, frame) -> None:
"""Per-gene guide agreement, ordered by gene p."""
try:
from ...guide_concordance import guide_support
except Exception:
return
try:
support = guide_support(frame)
except Exception:
self.support.set_frame(None)
self.agreement.set_support(None)
return
if support is None or not len(support):
self.support.set_frame(None)
self.agreement.set_support(None)
return
table = support.reset_index()
terms = self._gene_terms(frame)
table.insert(0, "feature",
[terms.get(str(gene)) for gene in table["gene"]])
def verdict(row):
"""Whether one row's guide support is sufficient to trust."""
if row["single_guide"]:
return "single guide -- gene p IS that guide's p"
if row["concordance"] < 0.6:
return "guides disagree in direction"
if row["n_guides_significant"] == 0:
return "agreement is the evidence"
return "supported"
table["verdict"] = table.apply(verdict, axis=1)
if self._level == "gene":
table = table[table["feature"].notna()].reset_index(drop=True)
if not len(table):
self.support.set_frame(None)
self.agreement.set_support(None)
self.agreement.set_status(
"No gene in this table was fitted a gene-level term, so "
"“genes only” leaves nothing to draw here. "
+ self.GUIDE_SUPPORT_NEEDS_BOTH)
return
usable = table["feature"].notna().all() and table["feature"].is_unique
self.support.set_frame(table,
key_column="feature" if usable else None)
self.agreement.set_support(table, keys=table["feature"])
[docs]
def ranking(self):
"""``(kind, column)`` for the table on screen.
``kind`` is ``"p-value"``, ``"selection-frequency"`` or ``None``.
The panel is handed coefficient tables from every backend spaCR can
fit and they do not agree on this: OLS reports ``P>|t|``, a GLM or a
Poisson fit ``P>|z|``, spaCR's own writer ``p_value``, and the
penalised backends report no p-value whatsoever.
"""
return self._ranking
@staticmethod
def _rank_by(frame, source: str = ""):
"""What orders this table, and which column carries it.
THE FOLDER OVERRULES THE COLUMNS on the penalised backends. spacr.ml
writes an OLS-style ``p_value`` into a lasso ``results.csv`` -- it is
computed as though there were no penalty, which is why
:data:`spacr.hits.NO_P_VALUE_TYPES` exists -- so a panel that trusts
the column draws a p-value histogram of a quantity that is not a
p-value, and a q-value would be a correction applied to it.
"""
selection = _match_column(frame, SELECTION_COLUMNS)
backend = backend_of(source)
if backend is not None:
return ("selection-frequency", selection)
if selection is not None:
return ("selection-frequency", selection)
p_column = _match_column(frame, P_VALUE_COLUMNS)
if p_column is not None:
return ("p-value", p_column)
return (None, None)
@staticmethod
def _p_column(frame) -> Optional[str]:
"""The p-value column, however this backend spelled it."""
return _match_column(frame, P_VALUE_COLUMNS)
def _show_significance(self, frame, kind, column, source: str = "") -> None:
"""Fill the p-value and Q-Q tabs, or say why they are empty.
An empty histogram with no caption reads as a broken panel. A
penalised fit has no p-value to draw and never will, and a table
carrying only a ``z value`` has a statistic rather than a test -- both
are answers, and both used to be silence.
"""
rows = len(frame)
where = source or "this table"
if kind == "p-value":
keys = self._keys_for(frame)
self.p_values.set_p_values(frame[column], keys=keys)
self.qq.set_p_values(frame[column], keys=keys)
self.say(f"{rows} coefficients from {where}, ranked by "
f"“{column}”.")
return
backend = backend_of(source)
if kind == "selection-frequency":
named = f"“{column}”" if column else "bootstrap selection frequency"
which = f"{backend} " if backend else ""
reason = (f"A {which}fit is ranked by bootstrap selection "
f"frequency ({named}) and has no p-value: it is a "
f"selection method, not a hypothesis test.")
else:
statistic = _match_column(frame, STATISTIC_COLUMNS)
carries = (f"It carries “{statistic}”, which is a test statistic "
f"and not a p-value. " if statistic else "")
reason = (f"No p-value column in this table. {carries}"
f"Looked for {', '.join(P_VALUE_COLUMNS[:4])} and the "
f"rest of the usual spellings.")
self.p_values.set_p_values([])
self.p_values.set_status(reason)
self.qq.set_p_values([])
self.qq.set_status(reason)
self.say(f"{rows} coefficients from {where}. {reason}")
@classmethod
def _keys_for(cls, frame):
"""The identifier of every row, in frame order, or ``None``.
``None`` is a real answer: a table with no column that names a row
uniquely has nothing to join on, and a plot given no keys stays
readable and simply does not claim its points are clickable. That
beats handing over ``gene``, which repeats across a gene's guides and
would select an arbitrary one of them.
"""
column = cls._key_column(frame)
return None if column is None else frame[column]
@staticmethod
def _effect_column(frame) -> str:
"""Find the column holding the effect size.
:param frame: the results table.
:returns: the first of the known effect column names present, falling
back to ``"coefficient"`` so a caller always has a name to use.
"""
for name in ("coefficient", "coef", "effect", "estimate"):
if name in frame.columns:
return name
return "coefficient"
#: Columns offered for colouring EVEN WHEN the generic filter would drop
#: them. A screen's own control annotation is worth seeing whatever it
#: contains -- a single value means the control names matched nothing,
#: which the user needs to know rather than to be shielded from.
ALWAYS_OFFERED = ("condition",)
#: Which rows EVERY tab shows, in the order every control offers them.
#:
#: gRNA FIRST, because it is what a new table opens on -- see
#: :meth:`_default_level` -- and a control whose first entry is not its
#: current value reads as a control that has been changed. "Both" LAST,
#: because it is the one that puts two multiple-testing families on one
#: axis and is therefore the one to choose deliberately.
LEVELS = (("grna", "guides only"), ("gene", "genes only"),
(None, "genes and guides"))
#: The level, as a control names it: the words the reader asks for. The
#: sentences in :data:`LEVEL_NAMES` say what each one DOES and are what
#: the status line and the menus use; a combo box beside four other
#: controls has room for a name and not for a sentence, and the two are
#: the same three values either way.
LEVEL_SHORT = {None: "Both", "gene": "Gene", "grna": "gRNA"}
#: The level, as it is said in a sentence.
LEVEL_NAMES = {None: "genes and guides", "gene": "genes only",
"grna": "guides only"}
#: The level, as it is appended to a tab. Short, because it goes on five
#: tab labels at once and a tab bar that wraps has hidden the filter it
#: was put there to advertise.
LEVEL_SUFFIXES = {None: "", "gene": " (genes)", "grna": " (guides)"}
#: The level, as it is appended to a PLOT's title. Longer than the tab
#: suffix: the title has the room, and a reader looking at a picture is
#: the reader most likely to forget which filter is on.
LEVEL_TITLES = {None: "", "gene": " — genes only",
"grna": " — guides only"}
#: Why the three well-level tabs do not follow the gene/guide filter.
#: SAID, not left to be noticed: a filter that reaches five tabs and
#: silently skips three is the same failure as one that reaches four of
#: six, only quieter.
DIAGNOSTICS_ARE_WHOLE_FIT = (
"NOT FILTERED: one point here is one WELL, and a well is neither a "
"gene nor a guide. These describe the whole fit whatever the "
"coefficient tabs are showing.")
#: Why the guide-support tab is narrowed differently from the rest.
GUIDE_SUPPORT_NEEDS_BOTH = (
"Computed from the whole table, always: a gene's concordance is how "
"its own guides agree, so the guide rows are what the number is made "
"of and a gene-only table has none of them. The filter narrows which "
"genes are LISTED, never how one was measured.")
#: Sentinel for the derived LOPIT colouring, which is not a frame column.
LOPIT_KEY = "\0lopit"
#: What the volcano can measure its effects from, and what each says.
BASELINES = (
(None, "zero (no dose-response)"),
("controls", "the non-targeting controls"),
)
def _offer_baselines(self) -> None:
"""Put the baselines on the volcano's right-click menu."""
chosen = self._baseline[0]
self.volcano.offer_baselines([
(label, (lambda k=kind: self.set_baseline(k)), kind == chosen)
for kind, label in self.BASELINES])
[docs]
def both_levels_note(self) -> str:
"""One sentence naming the level shown and the one that is not.
THE RUN FITS TWICE AND THE PANEL SHOWS ONE. The design splits
`level='both'` into a guide fit and a gene fit -- two tables, two
multiple-testing families -- and the panel opens on guides so a gene
is not drawn once per guide. Both of those are right.
Previously, nothing said so. In a representative GLM run, both fits
completed -- ``results_grna.csv`` had 15 rows and ``results_gene.csv``
had 5 -- while half of
it was invisible with nothing on screen naming the other half, so "it
only runs once" is the honest reading from the user's side.
Empty when there is nothing to say: one level in the table, or no
filter on. A note that fires every time is a note nobody reads.
A LEVEL WITH NO ROWS IS A DIFFERENT SENTENCE, and this one gives way
to it. "genes only: 0 of 789 coefficients. The guide fit is in this
run too" is true and is not an answer -- the reader is looking at an
empty tab and needs to know that this run HAS no gene fit and why.
See :meth:`missing_level_note`.
"""
missing = self.missing_level_note()
if missing:
return missing
counts = self.level_counts()
total = counts.get(None, 0)
if not self._level or not total:
return ""
shown = counts.get(self._level, 0)
other = "gene" if self._level == "grna" else "grna"
if not counts.get(other, 0):
return ""
return (f"{self.LEVEL_NAMES.get(self._level, 'everything')}: "
f"{shown} of {total} coefficients. The "
f"{'gene' if other == 'gene' else 'guide'} fit is in this run "
f"too — switch with Level.")
[docs]
def missing_level_note(self) -> str:
"""Explain why the selected model level contains no result rows.
Guide permutations and guide-only fits legitimately omit gene-level
rows. The returned note identifies the available level; an empty
string indicates that rows are present.
"""
if not self._level or self._frame is None:
return ""
counts = self.level_counts()
if counts.get(self._level, 0):
return ""
wanted = "gene" if self._level == "gene" else "guide"
other_key = "grna" if self._level == "gene" else "gene"
other = "guide" if other_key == "grna" else "gene"
note = (f"No {wanted}-level coefficients in this run: "
f"{self._why_no_rows_at(self._level)}")
if counts.get(other_key, 0):
note += (f" Its {counts[other_key]} {other}-level coefficients "
f"are still here — set Level to "
f"{self.LEVEL_SHORT[other_key]} to read them.")
return note
def _why_no_rows_at(self, level) -> str:
"""The reason this table holds no rows at ``level``, as a clause.
Read off the TABLE first and the settings second. A run folder does
not always keep its settings beside the results -- neither of the two
real runs this was measured on did -- so a reason that depended on
them would be absent exactly when the panel is opened from disk,
which is the case that needs it.
"""
frame = self._frame
columns = list(getattr(frame, "columns", ()))
settings = self._run_settings if isinstance(self._run_settings,
dict) else {}
fitted = str(settings.get("level") or "").strip().lower()
if level == "gene":
if any(name in columns for name in PERMUTATION_COLUMNS):
return ("this permutation run reported guides only. The gene "
"pass tests each gene as a SET -- its regressor the "
"sum of its guides' fractions, permuted with the same "
"scheme and corrected as its own family -- and it "
"runs when level is 'gene' or 'both'. Re-run at one "
"of those to get gene rows.")
if fitted == "grna":
return ("it was fitted at level='grna', which fits the guide "
"terms only. Re-fit at level='both' or 'gene' to get "
"gene coefficients.")
if "level" in columns:
return ("every row it wrote comes from the guide fit — the "
"table records the fit each coefficient came from, "
"and none of them says gene.")
return ("no term in the coefficient table names a gene. Re-fit "
"at level='both' or 'gene' to get gene coefficients.")
if fitted == "gene":
return ("it was fitted at level='gene', which fits the gene "
"terms only. Re-fit at level='both' or 'grna' to get "
"per-guide coefficients.")
return ("no term in the coefficient table names a guide — which is "
"what the separate gene fit writes.")
[docs]
def level_counts(self) -> dict:
"""``{None: n, "gene": n, "grna": n}`` for the table on screen.
Counted rather than assumed, and put in the MENU: "genes only" that
silently draws 400 of 1,213 points is a filter a user applies without
knowing what they gave up.
"""
frame = self._frame
levels = self._levels_of(frame)
if frame is None or levels is None:
return {None: 0, "gene": 0, "grna": 0}
return {None: int(len(frame)),
"grna": int((levels == "grna").sum()),
"gene": int((levels == "gene").sum())}
def _default_level(self):
"""The level to open a new table at.
Guides when the table has any -- the guide is the unit the screen
measures, and drawing a gene once per guide is the duplication this
default exists to prevent. Genes when it has no guide terms at all,
which is exactly what the separate gene fit writes. Whole fit when it
has neither, so an unrecognised table is shown rather than hidden.
"""
counts = self.level_counts()
if counts.get("grna", 0):
return "grna"
if counts.get("gene", 0):
return "gene"
return None
def _offer_levels(self) -> None:
"""Synchronize coefficient-level choices across all result views.
The panel selector, volcano control, and coefficient-table menu are
rebuilt from :attr:`_level`, ensuring that each control displays the
coefficient level currently rendered by the panel.
The level summary uses the plot's persistent option-note slot so point
selections cannot replace the explanation of hidden coefficients.
"""
counts = self.level_counts()
note = self.both_levels_note()
self.volcano.offer_levels(
[(f"{label} ({counts.get(key, 0)})",
(lambda k=key: self.set_level(k)), key == self._level)
for key, label in self.LEVELS],
note=note)
blocked = self._level_box.blockSignals(True)
self._level_box.clear()
for key, label in self.LEVELS:
self._level_box.addItem(
f"{self.LEVEL_SHORT.get(key, label)} ({counts.get(key, 0)})",
key)
index = self._level_box.count() - 1
self._level_box.setItemData(index, label, Qt.ToolTipRole)
chosen = self._level_box.findData(self._level)
self._level_box.setCurrentIndex(max(chosen, 0))
self._level_box.blockSignals(blocked)
missing = self.missing_level_note()
self._missing_level.setText(missing)
self._missing_level.setVisible(bool(missing))
def _level_chosen(self, index: int) -> None:
"""The reader picked a level off the panel's own control."""
if 0 <= int(index) < self._level_box.count():
self.set_level(self._level_box.itemData(int(index)))
def _level_menu_at(self, position) -> None:
"""Open the coefficient-level menu for the results table.
The table and volcano menus update the same level state so every
downstream table and plot uses the same gene/guide filter.
"""
self.build_level_menu().exec(
self.table.table.viewport().mapToGlobal(position))
[docs]
def level(self):
"""``None``, ``"gene"`` or ``"grna"`` -- which family every tab shows.
Public because it is the one piece of state that decides what SIX
tabs are drawing, and a caller that has to read it off a private
attribute is a caller that will one day set it there too.
"""
return self._level
[docs]
def filtered_frame(self):
"""The coefficient table at the chosen level, or ``None``.
This is what every tab is drawn from -- see :meth:`refresh_views`.
:meth:`results_frame` is the RUN's table and stays whole: the filter
is a view, not an edit, and a caller exporting the results must get
the fit rather than whatever the user last right-clicked.
"""
frame = self._frame
if frame is None:
return None
mask = self._level_mask(frame)
return frame if mask is None else frame.loc[mask]
[docs]
def family_note(self) -> str:
"""Which multiple-testing family the p-value and Q-Q tabs are drawing.
THE ONE THING FILTERING A Q-Q CHANGES THAT FILTERING A VOLCANO DOES
NOT. A volcano of the guides is the same dots with some removed; a
Q-Q of the guides is a DIFFERENT DIAGNOSTIC -- the expected quantiles
are recomputed over 900 tests instead of 1,200, so the diagonal moves,
the inflation figure at the median is a different number, and the
excess in the histogram's first bin is this family's excess and not
the run's. A reader who does not know which one is on screen cannot
use either.
"""
counts = self.level_counts()
total = counts.get(None, 0)
if not self._level:
return (f"Every tab covers the whole fit: all {total} "
f"coefficients, one multiple-testing family.")
missing = self.missing_level_note()
if missing:
return missing
family = "genes" if self._level == "gene" else "guides"
return (f"{family} only — {counts.get(self._level, 0)} of {total} "
f"coefficients. The p-value histogram and the Q-Q are "
f"therefore a DIFFERENT multiple-testing family from the "
f"whole fit: the inflation figure and the excess in the "
f"first bin are this family's, not the run's.")
[docs]
def set_level(self, level) -> None:
"""Draw only genes, only guides, or both -- ON EVERY TAB.
A FILTER ON THE PANEL, not a mode of the run. The coefficient table
already carries both -- `feature` is `gene_fraction:gene[...]` or
`fraction:grna[...]` -- so this needs no re-fit and no second table,
which is why it is better here than in the settings where it used to
live.
ONE PIECE OF STATE. Reached from the volcano's right-click menu and
from the coefficients table's, and read by every draw path, because a
filter that reaches four of six tabs is worse than one that reaches
none: the two then disagree on screen at the same time and nothing
says which is which.
:param level: ``"gene"`` for genes only, ``"grna"`` for guides only,
or ``None`` for both.
"""
self._level = level
self._offer_levels()
self.refresh_views()
self._mark_the_level_on_the_plots()
counts = self.level_counts()
shown = counts.get(level, counts.get(None, 0))
missing = self.missing_level_note()
headline = missing or (
f"{shown} of {counts.get(None, 0)} coefficients — "
f"{self.LEVEL_NAMES.get(level, 'genes and guides')}, on "
f"every tab. {self.family_note()}")
self.say(headline,
detail=f"{self.family_note()}\n\n"
f"{self.DIAGNOSTICS_ARE_WHOLE_FIT}\n\n"
f"{self.GUIDE_SUPPORT_NEEDS_BOTH}")
def _filtered_surfaces(self) -> tuple:
"""``[(tab widget, tab name, plot, plot title)]`` the filter narrows.
The family is written into BOTH the tab and the plot's own title: a
tab label survives a click where a plot's status line does not, and
the title is what a reader is looking at when they read the picture.
"""
return (
(self._volcano_tab, self._volcano_tab_name, self.volcano,
"Volcano"),
(self.effect_rank, "Effect rank", self.effect_rank,
"Effect rank"),
(self.effect_distribution, "Effect distribution",
self.effect_distribution, "Effect distribution"),
(self.p_values, "p-values", self.p_values,
"p-value distribution"),
(self.qq, "Q-Q", self.qq, "p-value Q-Q"),
(self.controls, "Controls", self.controls, "Control separation"),
(self._support_tab, "Guide support", self.agreement,
"Guide agreement"),
)
def _say_which_family(self) -> None:
"""Write the level onto every tab and every plot that follows it."""
suffix = self.LEVEL_SUFFIXES.get(self._level, "")
note = self.family_note()
for widget, name, plot, title in self._filtered_surfaces():
index = self.tabs.indexOf(widget)
if index >= 0:
self.tabs.setTabText(index, f"{name}{suffix}")
self.tabs.setTabToolTip(index, note)
plot.plot.setTitle(
f"{title}{self.LEVEL_TITLES.get(self._level, '')}")
support = self.tabs.indexOf(self._support_tab)
if support >= 0:
self.tabs.setTabToolTip(
support, f"{note}\n\n{self.GUIDE_SUPPORT_NEEDS_BOTH}")
self._note_the_diagnostics()
def _note_the_diagnostics(self) -> None:
"""Say on the well-level tabs that the filter does not reach them.
They are one point per WELL. There is no gene/guide split to make and
pretending to make one would be worse than the silence it replaces --
but silence is what let five tabs narrow while three did not, with
nothing on screen distinguishing "unfiltered on purpose" from
"forgot to filter".
"""
note = self.DIAGNOSTICS_ARE_WHOLE_FIT if self._level else ""
for plot in self.diagnostic_plots():
plot.set_status_note(note)
@staticmethod
def _levels_of(frame):
"""Which fit each row came from, asked of :mod:`spacr.hits`.
ONE STATEMENT OF THE SPLIT, asked rather than re-derived. The mask,
the counts and the opening level were three separate parses of
``feature``, and the complement they used files the GUIDE fit's
``Intercept`` under "gene": a run at ``level='both'`` fits twice and
both fits report an intercept, so on the real screen "genes only" drew
382 rows where the run recorded 381. The table's own ``level`` column
is what tells the two apart.
"""
try:
from ...hits import coefficient_levels
except Exception:
return None
return coefficient_levels(frame)
def _level_mask(self, frame):
"""Rows of ``frame`` at the chosen level, or None for all of them."""
if not self._level or frame is None:
return None
columns = getattr(frame, "columns", ())
if "feature" not in columns and "level" not in columns:
return None
levels = self._levels_of(frame)
if levels is None:
return None
return levels == self._level
def _offer_p_values(self) -> None:
"""Offer raw and adjusted p-values when correction was performed.
A copied ``q_value`` from ``multiple_testing_method='none'`` does not
count as an adjusted value and does not enable the choice.
"""
frame = self._frame
corrected = (_match_column(frame, ("q_value", "adjusted_p_value"))
if frame is not None else None)
if corrected is None:
self.volcano.offer_p_values([])
return
method = ""
if "multiple_testing_method" in getattr(frame, "columns", ()):
values = frame["multiple_testing_method"].dropna().unique()
method = str(values[0]) if len(values) else ""
if method.lower() in ("none", "nan", ""):
self.volcano.offer_p_values([])
self._p_value_note = (
f"No correction was applied, so {corrected!r} equals the raw "
f"p-value; there is nothing to switch between.")
return
self._p_value_note = ""
self.volcano.offer_p_values([
(f"raw p-value", lambda: self.set_p_value_kind("raw"),
self._p_value_kind == "raw"),
(f"adjusted ({method})",
lambda: self.set_p_value_kind("adjusted"),
self._p_value_kind == "adjusted"),
])
[docs]
def set_p_value_kind(self, kind) -> None:
"""Draw the volcano against the raw or the adjusted p-value.
:param kind: ``"raw"`` or ``"adjusted"``.
"""
self._p_value_kind = kind
self._offer_p_values()
self._redraw_volcano()
self.say(f"Volcano y-axis: {kind} p-value."
+ (f" {self._p_value_note}" if self._p_value_note else ""))
def _offer_thresholds(self) -> None:
"""Add effect-size threshold controls to the volcano plot menu."""
from ...thresholds import METHODS
self.volcano.offer_thresholds(
[(f"{name} — {METHODS[name][1].split(' --')[0]}",
(lambda n=name: self.set_threshold_method(n)),
name == self._threshold_method) for name in METHODS],
multiplier=self._threshold_multiplier,
on_multiplier=self.set_threshold_multiplier)
[docs]
def set_threshold_method(self, method) -> None:
"""Measure the effect-size cut a different way, and redraw.
:param method: a key of :data:`spacr.thresholds.METHODS`, e.g.
``"none"`` or ``"std"``.
"""
self._threshold_method = method
self._offer_thresholds()
self._redraw_volcano()
self.say(self._threshold_sentence())
[docs]
def set_threshold_multiplier(self, multiplier) -> None:
"""How many spreads wide the cut is.
:param multiplier: number of spreads, converted to ``float``.
"""
self._threshold_multiplier = float(multiplier)
self._offer_thresholds()
self._redraw_volcano()
self.say(self._threshold_sentence())
def _current_threshold(self):
"""The effect-size cut for the table on screen, or None.
Split from :meth:`_threshold_sentence` so the number the plot draws
and the number the sentence quotes come from ONE call -- a panel
whose line and caption disagreed would be worse than one with no
line at all.
"""
from ...thresholds import coefficient_threshold
controls = self._control_effects()
if controls is None:
return None
value, _rule = coefficient_threshold(
controls, self._threshold_method, self._threshold_multiplier)
return value
def _control_effects(self):
"""The control guides' effects, or ``None`` when there are none to take.
THE EFFECT COLUMN IS CHECKED, NOT ASSUMED, and that is a crash rather
than tidiness. :meth:`_effect_column` answers ``"coefficient"`` for a
table carrying none of the four spellings -- it is a default, not a
finding -- and ``frame.loc[mask, name]`` on an absent column raises
KeyError. That escaped through :meth:`refresh_views` into
:meth:`set_frame`, so a table with a ``condition`` column and an
unrecognised effect column took the WHOLE PANEL down at load time
instead of opening and saying which column it could not find.
"""
frame = self._frame
if frame is None or "condition" not in getattr(frame, "columns", ()):
return None
effect = self._effect_column(frame)
if effect not in frame.columns:
return None
return frame.loc[
frame["condition"].astype(str).str.lower().isin(("nc", "control")),
effect]
def _threshold_sentence(self) -> str:
"""What the cut is, and what drew it. Never a bare number."""
from ...thresholds import coefficient_threshold, describe
controls = self._control_effects()
if controls is None:
return "No control coefficients, so no effect-size cut."
value, rule = coefficient_threshold(
controls, self._threshold_method, self._threshold_multiplier)
if value is None:
return f"No effect-size cut: {rule}."
return (f"Effect-size cut {value:.3g} — {rule}. "
f"{describe(self._threshold_method)}.")
def _offer_compartments(self) -> None:
"""Put this screen's own compartments on the volcano's menu.
NOT ALL 27 IN THE REFERENCE TABLE. A menu offering 22 choices that
would colour nothing is a menu where a choice that colours nothing is
indistinguishable from a broken one.
"""
from ...localisation import present
try:
names = present(self._frame) if self._frame is not None else []
except Exception: # noqa: BLE001
names = []
from ...localisation import ALL as ALL_COMPARTMENTS
options = [("none (up / down)", lambda: self.set_compartment(None),
self._compartment is None)]
options.append(("all localisations",
lambda: self.set_compartment(ALL_COMPARTMENTS),
self._compartment == ALL_COMPARTMENTS))
options += [(name, (lambda n=name: self.set_compartment(n)),
name == self._compartment) for name in names]
self.volcano.offer_compartments(options if names else [])
[docs]
def set_compartment(self, name) -> None:
"""Colour one TAGM/LOPIT compartment against grey, or none.
ONE. "Everything is grey except what the sentence is about" -- and a
27-colour volcano is both what that rule forbids and, measured, the
version whose legend cost 40 ms of a 49 ms redraw.
:param name: TAGM/LOPIT compartment name to pick out,
:data:`spacr.localisation.ALL` to colour every annotated
coefficient by its compartment, or ``None``/empty for none.
"""
self._compartment = name
self._offer_compartments()
self._redraw_volcano()
from ...localisation import ALL as ALL_COMPARTMENTS
if name == ALL_COMPARTMENTS:
from ...localisation import of as compartment_of
annotated = 0
total = 0
if self._frame is not None:
names = compartment_of(self._frame)
total = len(self._frame)
annotated = int((names.astype(str) != "").sum()) if len(names) else 0
self.say(
f"{annotated} of {total} coefficients carry a TAGM/LOPIT "
f"localisation and each is coloured by it; the rest are one "
f"colour marked 'elsewhere'."
if annotated else
"No coefficient in this screen carries a TAGM/LOPIT "
"localisation, so there is nothing to colour by.")
elif name:
from ...localisation import mask
found = int(mask(self._frame, name).sum()) if self._frame is not None else 0
self.say(f"{found} of {len(self._frame)} coefficients are "
f"annotated {name} in the TAGM/LOPIT table; the rest are "
f"grey."
if found else
f"No coefficient in this screen is annotated {name}, so "
f"nothing is picked out.")
[docs]
def set_baseline(self, kind, name=None) -> None:
"""Measure every effect from ``kind`` -- see :mod:`spacr.baseline`.
The interactive volcano only. The saved figures take their own
baseline argument, so a user who moved it here and then exported gets
a picture and a caption that agree.
:param kind: baseline kind from :mod:`spacr.baseline`: ``"zero"``,
``"controls"``, ``"named"`` or ``"value"``; ``None`` or empty means
zero.
"""
self._baseline = (kind, name)
self._offer_baselines()
self._redraw_volcano()
if self._frame is not None:
from ...baseline import resolve
chosen = resolve(self._frame, kind or "zero",
column=self._effect_column(self._frame),
name=name)
self.say(chosen.sentence
+ (f" Asked for the {kind} baseline, but "
f"{chosen.reason}." if chosen.reason else ""))
def _redraw_volcano(self) -> None:
"""Redraw the volcano, naming what its horizontal axis actually is.
The permutation path copies its partial correlation into
``coefficient`` so the rest of the screen can read one name, which
leaves the axis calling a bounded correlation a coefficient. It is named
per redraw rather than at load, because a panel can be handed a
different run without being rebuilt.
"""
if self._frame is None:
return
try:
self.volcano.name_the_effect(self._analysis_path())
except AttributeError: # noqa: BLE001
pass
kind, column = self._ranking
p_column = column if kind == "p-value" else "\0no p-value"
if kind == "p-value" and self._p_value_kind == "adjusted":
corrected = _match_column(self._frame,
("q_value", "adjusted_p_value"))
if corrected is not None:
p_column = corrected
from ...baseline import apply as apply_baseline
from ...baseline import resolve as resolve_baseline
effect_column = self._effect_column(self._frame)
baseline_kind, baseline_name = self._baseline
baseline = resolve_baseline(self._frame, baseline_kind or "zero",
column=effect_column, name=baseline_name)
frame = apply_baseline(self._frame, baseline, column=effect_column)
category = self._colour_by.currentData()
if category == self.LOPIT_KEY:
from ...localisation import of as compartments_of
frame = frame.copy()
frame["localisation"] = compartments_of(frame).replace(
"", "unannotated")
category = "localisation"
mask = self._level_mask(frame)
if mask is not None:
frame = frame.loc[mask]
drawn = self.volcano.set_results(
frame,
effect=effect_column,
p_column=p_column,
label_column="feature" if "feature" in frame.columns
else frame.columns[0],
category_column=category,
symbol_column=self._extra_colour_column(self._colour_by_2,
category),
opacity_column=self._extra_colour_column(self._colour_by_3,
category),
key_column=self._key_column(frame),
compartment=self._compartment,
effect_threshold=self._current_threshold(),
)
if not drawn:
self.volcano.set_keys(())
if kind != "p-value":
self.volcano.set_status(
"No p-value in this table, so there is no -log10(p) to plot. "
"The coefficients are in the Coefficients tab.")
if self._selected_key is not None:
self.volcano.highlight_key(self._selected_key)
#: Message shown in effect tabs before a coefficient table is available.
NO_EFFECTS_YET = (
"No coefficient table yet. Both effect tabs are drawn from the fitted "
"effects: run a regression, or open one with “Load results…”.")
#: Why these two tabs cannot be drawn from the table that IS loaded.
#: Named rather than shrugged at, because it is a specific and checkable
#: fact about the file: a frame with no fitted-effect column is not a
#: coefficient table, and no amount of re-running will make it one.
NO_EFFECT_COLUMN = (
"No fitted-effect column in this table, so there is nothing to rank "
"and nothing to distribute. Looked for coefficient, coef, effect and "
"estimate. Every other tab here is drawn from the same column, so a "
"table without it is not a regression result.")
def _draw_effects(self, frame, kind=None) -> None:
"""Fill the effect-rank and effect-distribution tabs.
Parameters
----------
frame : pandas.DataFrame
Coefficient table at the selected gene or guide level.
kind : str or None, optional
Ranking type returned by :meth:`ranking`. Only ``"p-value"``
permits significance coloring; penalized rankings must not reuse
inferential p-values computed under a different model.
Notes
-----
Unsupported plots display their reason in the tab. Nuisance terms are
excluded from both views because they are not part of the multiple-
testing family represented by the reported q-values.
"""
from .fast_plots import NO_SIGNIFICANCE
effect = self._effect_column(frame)
if effect not in getattr(frame, "columns", ()):
self.effect_rank.set_results(None)
self.effect_rank.set_status(self.NO_EFFECT_COLUMN)
self.effect_distribution.set_effects([])
self.effect_distribution.set_status(self.NO_EFFECT_COLUMN)
return
self.effect_rank.set_results(
frame, effect=effect, key_column=self._key_column(frame),
significance_column=(None if kind == "p-value"
else NO_SIGNIFICANCE))
tested = self._tested_mask(frame)
family = frame if tested is None else frame.loc[tested]
self.effect_distribution.set_effects(
family[effect], keys=self._keys_for(family),
untested=0 if tested is None else int((~tested).sum()))
@staticmethod
def _tested_mask(frame):
"""Rows of ``frame`` that are hypotheses, or ``None`` if it cannot say.
Via :func:`spacr.hits.tested_family`, which is the repository's single
statement of where the covariates end and the hypotheses begin. A
second copy of that rule is how the volcano, the sheet and this panel
would come to draw three different families.
"""
if "feature" not in getattr(frame, "columns", ()):
return None
try:
from ...hits import tested_family
except Exception:
return None
return tested_family(frame["feature"])
def _draw_controls(self, frame) -> None:
"""Split the effects by the control labels the fit assigned."""
if "condition" not in frame.columns:
self.controls.set_groups({})
return
effect = self._effect_column(frame)
if effect not in frame.columns:
self.controls.set_groups({})
return
groups, keys = {}, {}
key_column = self._key_column(frame)
names = {"nc": "negative", "pc": "positive", "control": "control",
"other": "other"}
for key, label in names.items():
rows = frame[frame["condition"].astype(str) == key]
if len(rows):
groups[label] = rows[effect].to_numpy()
if key_column is not None:
keys[label] = rows[key_column].astype(str).tolist()
self.controls.set_groups(groups, keys=keys or None)
@staticmethod
def _key_column(frame) -> Optional[str]:
"""The column that names a row uniquely, or ``None`` if none does.
Checked rather than assumed. ``feature`` is the design-matrix term
name and is one-to-one with the row on every table this module writes,
but a frame arriving from somewhere else may not carry it -- and
``gene`` and ``grna`` are NOT keys, because a gene has several guides
and therefore several rows. Joining on one of those selects an
arbitrary member of the group.
"""
if frame is None:
return None
for name in ("feature",):
if name in frame.columns and frame[name].is_unique:
return name
return None
def _select_from_a_plot(self, key: str) -> None:
"""Select a clicked coefficient, changing level when required.
Gene-only plots can be activated while the panel is filtered to
guides. In that case the level changes first and selection is applied
on the next event-loop turn, after the view rebuild and signal dispatch
finish. Clicks whose row is already visible do not change the filter.
"""
if self._reachable(key):
self.table.select_key(key)
return
from ...hits import guide_of
from PySide6.QtCore import QTimer
self.set_level("grna" if guide_of(str(key)) else "gene")
QTimer.singleShot(0, lambda k=str(key): self.table.select_key(k))
self.say(f"Showing {self.LEVEL_NAMES.get(self._level, 'everything')} "
f"so the point you clicked has a row.")
def _reachable(self, key) -> bool:
"""Whether ``key`` has a row in the table at the current level.
True when there is no filter, no frame or no `feature` column -- in
each of those the table is not hiding anything and an unfound key is
somebody else's problem to report.
"""
if not key or not self._level or self._frame is None:
return True
if "feature" not in getattr(self._frame, "columns", ()):
return True
mask = self._level_mask(self._frame)
if mask is None:
return True
features = self._frame["feature"].astype(str)
if str(key) not in set(features):
return True
return str(key) in set(features[mask])
def _select_key(self, key: str) -> None:
"""A row was picked: mark it on EVERY plot that drew it.
The link runs both ways on all of them, not just the volcano. A guide
found in the table should light up in the Q-Q as well -- that is how
a user answers "is my hit the one lifting off the diagonal", which is
the question the Q-Q exists for and could not previously be asked.
Each plot answers for itself, and a False is a real answer: a
coefficient with an unusable p-value is on no plot, a nuisance term is
off the volcano on purpose, and a guide is not a point on the
per-gene agreement plot at all.
"""
self._selected_key = str(key)
for plot in self._keyed_plots():
plot.note_selection(key, plot.highlight_key(key))
for histogram in (self.p_values, self.effect_distribution):
histogram.highlight_key(key)