"""One button: every gene against every measurement.
The engine is :mod:`spacr.gene_measurement_sweep`; this is
the thin panel over it, and it is thin on purpose -- the three corrections
that make the answer trustworthy (identifiers excluded, Benjamini-Hochberg
across the grid, circularity reported per row) live in the engine so a
settings CSV, a macro and this button cannot disagree about them.
IT RUNS OFF THE GUI THREAD, and the reason is today's crash rather than
politeness: a regression built Qt widgets on its own worker and the process
segfaulted somewhere else entirely. So the worker here touches NO widget and
returns plain data, and the figure is rendered by matplotlib's Agg canvas
which has no Qt object to own.
"""
from __future__ import annotations
import logging
import os
from typing import Any, Callable, Dict, Optional
import pandas as pd
from PySide6.QtCore import Qt, Signal
from PySide6.QtWidgets import (QAbstractItemView, QCheckBox, QComboBox,
QDoubleSpinBox, QFileDialog, QHBoxLayout,
QLabel, QLineEdit, QPushButton, QTableWidget,
QTableWidgetItem, QVBoxLayout, QWidget)
from .sortable_table import install_sorting, table_item
LOG = logging.getLogger("spacr.qt.sweep_panel")
def _names(text) -> list:
"""A comma-separated box as a list of names, or ``[]``.
Empty means "exclude nothing", and the caller turns that into an ABSENT
keyword rather than an empty list: `sweep` distinguishes "no filter" from
"a filter that matches nothing", and the second would be a silent way to
keep everything while looking like a choice.
"""
return [part.strip() for part in str(text or "").split(",")
if part.strip()]
__all__ = ["SweepPanel", "sweep_inputs"]
[docs]
class SweepPanel(QWidget):
"""The button, the table, and the picture.
The providers are CALLED when a sweep runs rather than read at
construction, so the panel always sweeps what is on screen now.
:param cells_provider: called for the cell table to sweep.
:param counts_provider: called for the per-well counts.
:param parent: parent widget.
:param threaded: whether the sweep runs off the GUI thread. False runs it
inline, which is what a test wants.
:param scores_provider: called for the scores, when the sweep needs them.
"""
finished = Signal(object)
def __init__(self, cells_provider: Optional[Callable] = None,
counts_provider: Optional[Callable] = None,
parent: Optional[QWidget] = None, *, threaded: bool = True,
scores_provider: Optional[Callable] = None):
"""Build the gene-sweep panel.
:param cells_provider: called for the per-object measurements.
:param counts_provider: called for the well or guide counts.
:param parent: parent widget, or ``None``.
:param threaded: run the sweep on a worker thread. Set ``False`` in
tests so ``run`` finishes before it returns.
:param scores_provider: called for the per-object scores. Separate from
``cells_provider`` because the merged measurements frame has no
prediction column -- it is the measurement tables -- so without this
circularity comes out NaN and the panel says so rather than showing
zeros.
"""
super().__init__(parent)
from ..job_runner import JobRunner
self._cells_provider = cells_provider
self._counts_provider = counts_provider
#: The per-object scores, so circularity can be computed at all. The
#: merged measurements frame has no `pred` column -- it is the
#: measurement tables -- so without this the column is NaN and the
#: panel says so rather than showing zeros.
self._scores_provider = scores_provider
self._result = None
#: Picture windows this panel opened, kept so Python does not collect
#: them the moment `show_picture` returns.
self._pictures: list = []
self._jobs = JobRunner(self, threaded=bool(threaded),
app_key="sweep every gene")
self._jobs.job_failed.connect(self._failed)
layout = QVBoxLayout(self)
row = QHBoxLayout()
self.run_button = QPushButton("Sweep every gene against every measurement")
self.run_button.setToolTip(
"Every guide against every measurement, blocked on plate and "
"corrected across the whole grid. Identifier columns are left "
"out. Each row also reports the absolute Spearman rank "
"correlation between the measurement and the classification "
"score, which identifies results that may reflect the score's "
"existing measurement basis.")
self.run_button.clicked.connect(self.start)
row.addWidget(self.run_button)
self._level_label = QLabel("rank by")
row.addWidget(self._level_label)
self.level = QComboBox()
for value, label in (("gene", "genes"), ("guide", "guides"),
("both", "genes and guides")):
self.level.addItem(label, value)
self._level_label.setToolTip(
"A gene's fraction in a well is the sum of its guide fractions, "
"matching the aggregation used by regression. Gene-level sweep "
"results therefore use the same guide aggregation as the fitted "
"gene effect.")
row.addWidget(self.level)
self._picture_label = QLabel("picture")
row.addWidget(self._picture_label)
self.picture = QComboBox()
for value, label in self.PICTURES:
self.picture.addItem(label, value)
self._picture_label.setToolTip(
"Which view of the sweep to draw and to save beside the table. "
"The heatmap summarizes surviving effects, calibration checks "
"the null p-value distribution, and the other views examine "
"effect size, hit counts, score correlation, guide prevalence, "
"profiles, similarity, or agreement among a gene's guides.")
row.addWidget(self.picture)
leave_out = QHBoxLayout()
leave_out.addWidget(QLabel("leave out — measurements"))
self.drop_columns = QLineEdit()
self.drop_columns.setPlaceholderText("column names, comma separated")
self.drop_columns.setToolTip(
"Measurement columns to keep out of the sweep, by name and comma "
"separated. Naming the two or three that are wrong is easier "
"than listing the seven hundred that are not.")
leave_out.addWidget(self.drop_columns, 1)
leave_out.addWidget(QLabel("genes or guides"))
self.drop_genes = QLineEdit()
self.drop_genes.setPlaceholderText("e.g. 220950, 233460")
self.drop_genes.setToolTip(
"Genes or guides to keep out, by name and comma separated. "
"Names are matched at both levels, so a gene identifier works whether the sweep is "
"running at gene or guide level.")
leave_out.addWidget(self.drop_genes, 1)
self.cap_wells = QCheckBox("drop guides in more than")
self.cap_wells.setToolTip(
"Leave out a guide present in more than this share of the wells. "
"Highly prevalent guides can have much greater precision than "
"rare guides and may be associated with many measurements. Use "
"the 'hits vs representation' view to choose a threshold from "
"this screen rather than assuming prevalence is harmless.")
leave_out.addWidget(self.cap_wells)
self.cap_wells_value = QDoubleSpinBox()
self.cap_wells_value.setDecimals(2)
self.cap_wells_value.setRange(0.05, 1.0)
self.cap_wells_value.setSingleStep(0.05)
self.cap_wells_value.setValue(0.5)
self.cap_wells_value.setSuffix(" of wells")
leave_out.addWidget(self.cap_wells_value)
leave_out.addStretch(1)
layout.addLayout(leave_out)
row.addWidget(QLabel("q <"))
self.alpha = QDoubleSpinBox()
self.alpha.setDecimals(3)
self.alpha.setRange(0.001, 0.5)
self.alpha.setSingleStep(0.01)
self.alpha.setValue(0.05)
self.alpha.valueChanged.connect(self._refill)
row.addWidget(self.alpha)
self.hide_circular = QCheckBox("hide what the score already tracks")
self.hide_circular.setToolTip(
"Hide rows whose measurement has an absolute Spearman rank "
"correlation of 0.15 or more with the classification score. "
"These rows may reflect the score's existing measurement basis "
"and are not independent corroboration.")
self.hide_circular.setChecked(True)
self.hide_circular.stateChanged.connect(self._refill)
row.addWidget(self.hide_circular)
row.addStretch(1)
self.show_button = QPushButton("Show picture")
self.show_button.setToolTip(
"Draw the chosen view of this sweep in its own window.")
self.show_button.clicked.connect(self.show_picture)
self.show_button.setEnabled(False)
row.addWidget(self.show_button)
self.save_button = QPushButton("Save table…")
self.save_button.clicked.connect(self.save)
self.save_button.setEnabled(False)
row.addWidget(self.save_button)
layout.addLayout(row)
self.status = QLabel("")
self.status.setWordWrap(True)
layout.addWidget(self.status)
self.table = QTableWidget(0, 8)
install_sorting(self.table)
self.table.setHorizontalHeaderLabels(
["level", "gene / guide", "measurement", "effect", "q",
"circularity", "wells", "effective"])
self.table.setSortingEnabled(True)
self.table.setEditTriggers(QAbstractItemView.NoEditTriggers)
self.table.setSelectionBehavior(QAbstractItemView.SelectRows)
layout.addWidget(self.table, 1)
from ..screens.settings_model import retarget_field_tooltips
retarget_field_tooltips(self)
[docs]
def start(self, *_args) -> bool:
"""Run the sweep. Returns whether one was started."""
if self._cells_provider is None or self._counts_provider is None:
self.status.setText(
"Nothing to sweep: this panel has no measurements and no "
"counts wired to it.")
return False
try:
cells = self._cells_provider()
counts = self._counts_provider()
except Exception as error: # noqa: BLE001
self.status.setText(f"Could not read the inputs: {error}")
return False
if cells is None or not len(cells) or counts is None or not len(counts):
self.status.setText(
"Nothing to sweep: merge the measurement databases first, and "
"attach the count CSVs the run was fitted on.")
return False
self.run_button.setEnabled(False)
self.status.setText("Sweeping…")
scores = None
if self._scores_provider is not None:
try:
scores = self._scores_provider()
except Exception: # noqa: BLE001
LOG.debug("could not read the scores", exc_info=True)
level = str(self.level.currentData() or "gene")
exclusions = self.exclusions()
return bool(self._jobs.submit(
lambda: self._work(cells, counts, scores, level, exclusions),
self._done))
@staticmethod
def _work(cells, counts, scores=None, level="gene", exclusions=None):
"""Run the sweep. Off the GUI thread.
:param cells: the per-object measurements.
:param counts: the well or guide counts.
:param scores: the per-object scores, or ``None``.
:param level: whether to sweep per gene or per guide.
:param exclusions: extra keyword arguments passed through to the sweep.
:returns: the sweep result.
"""
from ...gene_measurement_sweep import sweep
wells, fractions, plates, found = sweep_inputs(cells, counts,
scores=scores)
return sweep(wells, fractions, blocks=plates, scores=found,
level=level, **dict(exclusions or {}))
def _done(self, result) -> None:
"""Install a finished sweep and enable what it makes possible.
A sweep that returned nothing says so rather than leaving the buttons
armed over an empty result.
:param result: the sweep result, or ``None``.
"""
self.run_button.setEnabled(True)
self._result = result
if result is None:
self.status.setText("The sweep returned nothing.")
return
self.save_button.setEnabled(True)
self.show_button.setEnabled(True)
self.status.setText(result.describe())
self._refill()
self.finished.emit(result)
def _failed(self, message: str) -> None:
"""Re-enable Run and report a failed sweep.
:param message: the failure text from the job runner.
"""
self.run_button.setEnabled(True)
self.status.setText(f"The sweep did not finish: {message}")
[docs]
def rows(self) -> pd.DataFrame:
"""What the table is showing, as a frame."""
if self._result is None:
return pd.DataFrame()
bar = 0.15 if (self.hide_circular.isChecked()
and self._result.circularity_known) else 1.0
return self._result.survivors(alpha=float(self.alpha.value()),
max_circularity=bar)
def _refill(self, *_args) -> None:
"""Fill the results table, capped at the first 2,000 rows.
Sorting is switched off while rows are inserted: a sorted table
re-orders on every write, so the row index the next write uses is no
longer the row it just filled. When the cap bites, the status line says
so and points at the saved table for the rest.
:param _args: whatever the emitting control passes; ignored.
"""
keep = self.rows()
self.table.setSortingEnabled(False)
self.table.setRowCount(0)
shown = keep.head(2000)
self.table.setRowCount(len(shown))
for row, (_index, entry) in enumerate(shown.iterrows()):
values = [str(entry.get("level", "guide")), str(entry["guide"]),
str(entry["measurement"]),
f"{entry['effect']:+.3f}", f"{entry['q']:.2e}",
("—" if pd.isna(entry["circularity"])
else f"{entry['circularity']:.2f}"),
str(int(entry["n_wells"])),
f"{entry['effective_wells']:.0f}"]
for column, text in enumerate(values):
self.table.setItem(row, column, table_item(text))
self.table.setSortingEnabled(True)
if len(keep) > len(shown):
self.status.setText(
f"{self.status.text()} Showing the first {len(shown):,} of "
f"{len(keep):,} — save the table for all of them.")
[docs]
def exclusions(self) -> dict:
"""Return active exclusion controls as :func:`sweep` arguments.
Blank controls are omitted so they mean "no filter" rather than a
filter whose value matches nothing.
"""
out: dict = {}
columns = _names(self.drop_columns.text())
if columns:
out["drop_measurements"] = columns
genes = _names(self.drop_genes.text())
if genes:
out["drop_guides"] = genes
if self.cap_wells.isChecked():
out["max_wells_fraction"] = float(self.cap_wells_value.value())
return out
[docs]
def selected_gene(self):
"""Return the selected gene or the strongest surviving gene.
``None`` is returned when the table contains no suitable gene. The
profile title distinguishes the automatic fallback from a row selected
by the user.
"""
rows = {index.row() for index in self.table.selectedIndexes()}
if rows:
item = self.table.item(sorted(rows)[0], 1)
if item is not None and item.text().strip():
return item.text().strip()
if self._result is None or not len(self._result.table):
return None
best = self._result.table.sort_values("q")
return str(best.iloc[0]["guide"]) if len(best) else None
#: The pictures this panel can draw, in the order the chooser offers
#: them. Each answers a distinct question; see
#: :mod:`spacr.gene_measurement_sweep`. The heatmap remains first and is
#: therefore the default, while calibration is the first diagnostic to
#: inspect before interpreting the remaining views.
PICTURES = (
("heatmap", "what survived, clustered"),
("calibration", "is the screen calibrated at all?"),
("volcano", "every pair: effect against evidence"),
("representation", "hits vs how much screen the gene is"),
("families", "what KIND of measurement each gene moves"),
("profile", "one gene's fingerprint"),
("similarity", "which genes behave alike"),
("measurements", "which measurements discriminate"),
("circularity", "corroboration or restatement"),
("concordance", "do a gene's own guides agree?"),
)
[docs]
def show_picture(self, *_args):
"""Draw the chosen view in its own window. Returns the dialog, or None.
A WINDOW RATHER THAN A PANE, because the Measurements tab is already
four sections in a side panel -- "there are to many elements in the
measurements tab" -- and a heatmap of forty measurements needs more
width than that column has.
"""
from PySide6.QtWidgets import QDialog, QVBoxLayout
if self._result is None:
self.status.setText("Run the sweep first — there is nothing to "
"draw yet.")
return None
kind = str(self.picture.currentData() or "heatmap")
label = dict(self.PICTURES).get(kind, kind)
try:
figure = self.figure(kind=kind)
except Exception as exc: # noqa: BLE001
LOG.debug("could not draw the sweep picture", exc_info=True)
self.status.setText(f"That picture could not be drawn: {exc}")
return None
if figure is None:
self.status.setText(
f"Nothing to draw for “{label}” at q < "
f"{float(self.alpha.value()):g}. "
"Loosen the q filter, or pick another picture.")
return None
from .graph_builder import _canvas_class
dialog = QDialog(self)
dialog.setWindowTitle(f"Sweep — {label}")
layout = QVBoxLayout(dialog)
layout.setContentsMargins(4, 4, 4, 4)
canvas = _canvas_class()(figure)
layout.addWidget(canvas)
dialog.resize(1000, 720)
dialog.show()
self._pictures.append(dialog)
return dialog
[docs]
def save(self, *_args) -> str:
"""Write the whole table -- not the page on screen."""
if self._result is None:
return ""
path, _filter = QFileDialog.getSaveFileName(
self, "Save the sweep", "gene_measurement_sweep.csv",
"CSV (*.csv)")
if not path:
return ""
self._result.table.to_csv(path, index=False)
figure_path = os.path.splitext(path)[0] + ".png"
try:
self.figure(path=figure_path)
except Exception: # noqa: BLE001
LOG.debug("could not draw the sweep figure", exc_info=True)
self.status.setText(f"Saved {len(self._result.table):,} rows to {path}")
return path
[docs]
def closeEvent(self, event): # noqa: N802 - Qt name
"""Stop background work and unlink before going away.
:param event: the Qt close event.
"""
try:
self._jobs.shutdown()
finally:
super().closeEvent(event)