"""Pre-flight validation of a spaCR settings dict.
Every crash this module exists to prevent was, in practice, a twenty-line
check away: an ``organelle_channel`` one past the end of a three-channel
plate, a ``src`` that points at the plate folder instead of ``plate/merged``,
a ``cell_mask_dim`` beyond the last plane of the merged array, an integer
that came back from a settings CSV as the string ``"4"``. Each of those costs
a full GPU run to discover.
The module is deliberately dependency-light: it imports nothing from spaCR
except :mod:`spacr.settings` (which itself imports only ``os`` and ``ast``),
and it touches ``numpy`` only lazily, to read the *header* of a single
``.npy`` file. No torch, no cellpose, no image decoding. Importing and
running it costs well under a second, which is the whole point — it is what
``dry_run`` uses to answer "would this run work?" before anything is
allocated, loaded or written.
Public API
----------
``Problem``
One thing that is wrong, with the fix.
``validate_settings(settings, app_key)``
Returns a list of :class:`Problem`, errors and warnings mixed.
``format_report(problems, settings, app_key)``
Human-readable report; errors first, each with its fix line.
``describe_plan(settings, app_key)``
"Here is what would actually happen" summary.
Rules are derived from the code that consumes the settings, not invented:
each check names its source in a comment.
"""
from __future__ import annotations
import difflib
import os
import re
import shutil
import sys
from dataclasses import dataclass
from typing import Any, Dict, List, Optional, Sequence, Tuple, Union
from .object_roles import ALL_ROLES, SEGMENTED_ROLES
__all__ = [
"Problem",
"validate_settings",
"format_report",
"describe_plan",
"ERROR",
"WARNING",
"APP_FUNCTIONS",
]
ERROR = "error"
WARNING = "warning"
@dataclass
[docs]
class Problem:
"""One thing wrong with a settings dict.
:param severity: ``"error"`` (the run would fail or silently produce
wrong output) or ``"warning"`` (suspicious, but runnable).
:param setting: the settings key at fault, or ``""`` when the problem is
about the dataset rather than a single key.
:param message: what is wrong, phrased in the user's terms.
:param fix: what to actually do about it.
"""
severity: str
setting: str
message: str
fix: str
@property
[docs]
def is_error(self) -> bool:
"""True when this problem would break or corrupt the run."""
return self.severity == ERROR
[docs]
def __str__(self) -> str:
"""The problem and its fix, on two lines.
The setting's name leads when there is one, so a reader scanning a list of
problems sees WHICH setting each belongs to before the message.
"""
head = f"[{self.setting}] {self.message}" if self.setting else self.message
return f"{head}\n fix: {self.fix}"
APP_FUNCTIONS: Dict[str, str] = {
'host_pathogen': 'spacr.host_pathogen.analyze_host_pathogen',
"mask": "spacr.core.preprocess_generate_masks",
"timelapse": "spacr.core.preprocess_generate_masks_timelapse",
"motility": "spacr.timelapse.automated_motility_assay",
"measure": "spacr.measure.measure_crop",
"classify": "spacr.deep_spacr.deep_spacr",
"classify_merged": "spacr.classify.classify",
"activation": "spacr.deep_spacr.generate_activation_map",
"foreign": "spacr.foreign.import_project",
"external_masks": "spacr.external_masks.prepare_external_masks",
"align": "spacr.align.align_folder",
"ops": "spacr.ops_engine.run_ops",
"umap": "spacr.core.generate_image_umap",
"train_cellpose": "spacr.submodules.train_cellpose",
"ml_analyze": "spacr.ml.generate_ml_scores",
"cellpose_masks": "spacr.spacr_cellpose.identify_masks_finetune",
"map_barcodes": "spacr.sequencing.generate_barecode_mapping",
"regression": "spacr.ml.perform_regression",
"explain_cv": "spacr.surrogate.run_explain_cv",
"investigate_hit": "spacr.hit_investigation.investigate_hit",
"recruitment": "spacr.submodules.analyze_recruitment",
"invasion": "spacr.submodules.analyze_invasion",
"replication": "spacr.submodules.analyze_replication",
"endodyogeny": "spacr.submodules.analyze_endodyogeny",
"analyze_plaques": "spacr.submodules.analyze_plaques",
"convert": "spacr.convert.convert_folder",
"simulation": "spacr.sim.run_multiple_simulations",
"illumination": "spacr.illumination.prepare_illumination_correction",
"barcode_qc": "spacr.sequencing_qc.barcode_qc",
"anndata_export": "spacr.anndata_export.run_anndata_export",
}
APP_ALIASES: Dict[str, str] = {
"sequencing": "map_barcodes",
"barcodes": "map_barcodes",
"barcode_mapping": "map_barcodes",
"preprocess_generate_masks": "mask",
"generate_masks": "mask",
"measure_crop": "measure",
"deep_spacr": "classify",
"train": "classify",
"generate_image_umap": "umap",
"embedding": "umap",
"analyze_replication": "replication",
"analyze_endodyogeny": "endodyogeny",
"cellpose_all": "cellpose_masks",
}
try:
from .plugins import plugin_apps as _plugin_apps
for _plugin_app in _plugin_apps():
APP_FUNCTIONS.setdefault(
_plugin_app.key, _plugin_app.entrypoint.replace(":", ".")
)
for _alias in _plugin_app.aliases:
APP_ALIASES.setdefault(
_alias.strip().lower().replace("-", "_"), _plugin_app.key
)
except Exception:
pass
DB_APPS = frozenset({"umap", "ml_analyze", "regression", "recruitment", 'host_pathogen',
"activation", "classify", "classify_merged",
"invasion", "replication", "endodyogeny"})
MERGED_APPS = frozenset({"measure"})
MASK_APPS = frozenset({"mask", "timelapse"})
ALT_SRC_KEYS: Dict[str, str] = {
"foreign": "images",
"external_masks": "inputs",
"ops": "genotype_source",
}
CHANNEL_KEYS: Tuple[str, ...] = tuple(
f"{role}_channel" for role in SEGMENTED_ROLES)
MASK_DIM_KEYS: Tuple[str, ...] = tuple(
f"{role}_mask_dim" for role in SEGMENTED_ROLES)
#: Exactly the SEGMENTED roles -- cytoplasm has no channel to validate.
OBJECT_NAMES = SEGMENTED_ROLES
IMAGE_EXTENSIONS = (".tif", ".tiff", ".png", ".jpg", ".jpeg", ".bmp")
_EXT_SUFFIX = r"\.(?:tif|tiff|png|jpg|jpeg|bmp)$"
METADATA_REGEXES: Dict[str, str] = {
"cellvoyager": (
r"(?P<plateID>.*)_(?P<wellID>.*)_T(?P<timeID>.*)F(?P<fieldID>.*)"
r"L(?P<laserID>..)A(?P<AID>..)Z(?P<sliceID>.*)C(?P<chanID>.*)" + _EXT_SUFFIX
),
"cq1": (
r"W(?P<wellID>.*)F(?P<fieldID>.*)T(?P<timeID>.*)Z(?P<sliceID>.*)"
r"C(?P<chanID>.*)" + _EXT_SUFFIX
),
"auto": (
r"(?P<plateID>.*)_(?P<wellID>.*)_T(?P<timeID>.*)F(?P<fieldID>.*)"
r"L(?P<laserID>.*)C(?P<chanID>.*)" + _EXT_SUFFIX
),
}
def _add_the_rest_of_the_convention_table() -> None:
"""Put every other microscope convention into :data:`METADATA_REGEXES`.
The three above are spelled out because they are PINNED -- they mirror
``spacr.utils._get_regex``'s original arms character for character and
are not allowed to drift. Everything else lives in
``spacr.regex_infer._METADATA_CONVENTIONS`` and is copied here so that
this module does not become a second, staler list of what spaCR can
parse.
WITHOUT THIS THE WARNING LIES. ``_candidate_patterns`` tries the chosen
``metadata_type`` first and then sweeps the rest; a key it has never
heard of makes ``raw_channels`` None, and the run advice then tells a
user whose Opera Phenix plate parses perfectly that none of their files
match and that they should pick 'cellvoyager', 'cq1' or 'auto'.
``spacr.regex_infer`` imports nothing outside the standard library, so
this keeps the promise in this module's own header that it stays
dependency-light.
"""
from .regex_infer import (_METADATA_CONVENTIONS,
_metadata_pattern_any_extension)
extensions = tuple(suffix.lstrip(".") for suffix in IMAGE_EXTENSIONS)
for record in _METADATA_CONVENTIONS:
key = record["key"]
if key in METADATA_REGEXES or key == "custom":
continue
METADATA_REGEXES[key] = _metadata_pattern_any_extension(
key, extensions)
_add_the_rest_of_the_convention_table()
def _normalize_app(app_key: Any) -> str:
"""Canonicalize a caller-supplied app key; unknown keys pass through.
Plugin aliases are REGISTERED with ``-`` folded to ``_`` (see the
``plugin_apps`` loop above), so the lookup has to fold the caller's key
the same way or the fold only ever hides the spelling the plugin wrote
down. The exact key is tried first, so a table entry that genuinely
carries a hyphen still wins over its folded twin.
An UNKNOWN key passes through in the spelling it arrived in, not the
folded one: the fold exists to reach a registered alias, and rewriting a
key nothing matched would hand the caller back a name it never used.
"""
if not isinstance(app_key, str):
return ""
key = app_key.strip().lower()
if key in APP_ALIASES:
return APP_ALIASES[key]
folded = key.replace("-", "_")
return APP_ALIASES.get(folded, key)
#: The public name for :func:`_normalize_app`, for callers outside this module.
#:
#: THREE OF THEM WERE DOING IT BY HAND AND GETTING IT WRONG. `ports.py:468`,
#: `ports.py:556` and `chaining.py:427` each wrote `APP_ALIASES.get(key, key)`,
#: which skips the hyphen fold -- so a caller who writes `barcode-mapping`
#: reaches `map_barcodes` through `validate` and reaches nothing at all
#: through `module_ports`, whose own docstring promises "every alias
#: APP_ALIASES accepts".
#:
#: Exported rather than left private so the next caller has something to reach
#: for. A private helper that three modules need is a public one that has not
#: been named yet.
canonical_app_key = _normalize_app
_KNOWN_KEYS_CACHE: Optional[frozenset] = None
def _known_setting_keys() -> frozenset:
"""Every settings key spaCR knows about.
Built from ``expected_types``, ``tooltips`` and ``categories``, plus every
key any ``set_default_*`` / ``get_*_settings`` helper in
:mod:`spacr.settings` produces. The helpers are pure dict-fillers (the
whole sweep costs well under a millisecond), so calling them is cheaper
and far less rot-prone than maintaining a hand-written list.
"""
global _KNOWN_KEYS_CACHE
if _KNOWN_KEYS_CACHE is not None:
return _KNOWN_KEYS_CACHE
keys = {"hash_inputs"}
from . import settings as _settings
keys.update(getattr(_settings, "expected_types", {}))
keys.update(getattr(_settings, "tooltips", {}))
for group in getattr(_settings, "categories", {}).values():
if isinstance(group, (list, tuple, set)):
keys.update(k for k in group if isinstance(k, str))
import contextlib
import io as _io
from . import graph_types as _graph_types
buf = _io.StringIO()
reading = _graph_types._READ_THE_PREFERENCE_STORE.set(False)
try:
with contextlib.redirect_stdout(buf):
for name, fn in list(vars(_settings).items()):
if not callable(fn):
continue
if not (name.startswith("set_") or name.startswith("get_")
or name.startswith("default_")
or name.startswith("deep_")):
continue
try:
produced = fn({})
except Exception:
try:
produced = fn()
except Exception:
continue
if isinstance(produced, dict):
keys.update(k for k in produced if isinstance(k, str))
finally:
_graph_types._READ_THE_PREFERENCE_STORE.reset(reading)
_KNOWN_KEYS_CACHE = frozenset(keys)
return _KNOWN_KEYS_CACHE
@dataclass
class _Inventory:
"""What a single ``src`` actually contains, established without loading pixels."""
src: str = ""
exists: bool = False
is_dir: bool = False
is_db_file: bool = False
merged_dir: Optional[str] = None
merged_exists: bool = False
merged_files: int = 0
merged_note: Optional[str] = None
stack_dir: Optional[str] = None
stack_files: int = 0
array_planes: Optional[int] = None
array_source: str = ""
raw_channels: Optional[int] = None
raw_channel_ids: Tuple[str, ...] = ()
raw_evidence: str = ""
raw_files: int = 0
#: Which of src / src/orig / src/consolidated the raw files were found
#: in. Recorded so the resource card can size them without repeating
#: the search.
raw_dir: str = ""
fields: Optional[int] = None
regex_used: str = ""
db_path: Optional[str] = None
db_exists: bool = False
def _listdir(path: Optional[str]) -> List[str]:
"""``os.listdir`` without dot-files, returning [] instead of raising on a bad path.
Names that start with a dot are left out: a macOS ``._<name>``
AppleDouble sidecar keeps the ``.npy`` or ``.tif`` ending of the file it
shadows and is not an image or an array, so counting it doubles every
count the preflight prints, and a sorted listing puts it before the real
fields :func:`_peek_planes` samples.
:param path: the folder to list, or a falsy value.
:returns: the visible entry names in :func:`os.listdir` order, or an
empty list when ``path`` is falsy or cannot be listed.
"""
if not path:
return []
try:
return [name for name in os.listdir(path) if not name.startswith('.')]
except OSError:
return []
def _resolve_merged_dir(src: str) -> str:
"""Where measure_crop would look for the merged arrays.
Mirrors spacr.measure.measure_crop, which appends ``merged`` unless
``os.path.basename(src)`` already ends with it.
"""
if os.path.basename(os.path.normpath(src)).endswith("merged"):
return src
return os.path.join(src, "merged")
_PLATE_OUTPUT_FOLDERS = frozenset({
"merged", "measure", "measurements", "masks", "datasets", "data",
"settings", "plots", "results", "stack", "norm_channel_stack",
"channel_stack", "orig", "png", "crops", "figures"})
def _holds_npy(directory: str) -> bool:
"""Whether ``directory`` is a folder with at least one visible ``.npy``."""
return os.path.isdir(directory) and any(
name.endswith(".npy") for name in _listdir(directory))
def _resolve_measure_src(src: str) -> Tuple[str, Optional[str]]:
"""The merged folder measure reads for ``src``, and why it was moved.
A plate root or a ``merged`` folder is read as it always was. When
``src`` names one of spaCR's own output subfolders of a plate
(``measurements``, ``measure``, ``masks``, ...) and that plate's
``merged`` folder holds ``.npy`` arrays, measure reads that folder
instead, whether or not the subfolder itself exists.
:param src: the ``src`` setting, one path.
:returns: ``(merged_folder, note)``. ``note`` is None when nothing was
moved, otherwise one sentence saying what was used instead.
"""
path = os.path.normpath(src)
direct = _resolve_merged_dir(path)
if _holds_npy(direct):
return direct, None
name = os.path.basename(path)
if name.lower() in _PLATE_OUTPUT_FOLDERS:
candidate = os.path.join(os.path.dirname(path), "merged")
if os.path.normpath(candidate) != os.path.normpath(direct) and _holds_npy(candidate):
return candidate, (
f"src points at {path}, the plate's '{name}' output folder; "
f"measure reads the merged arrays in {candidate} instead.")
return direct, None
def _peek_planes(directory: str) -> Tuple[Optional[int], str, int]:
"""Read the shape of ONE ``.npy`` in ``directory`` without loading its data.
:returns: ``(planes, example_filename, npy_count)``. ``planes`` is the
length of the last axis (1 for a 2-D array), or None when nothing
could be read.
"""
npys = sorted(f for f in _listdir(directory) if f.endswith(".npy"))
if not npys:
return None, "", 0
import numpy as np
for name in npys[:3]:
try:
arr = np.load(os.path.join(directory, name), mmap_mode="r")
except (OSError, ValueError, EOFError):
continue
shape = tuple(getattr(arr, "shape", ()))
if not shape:
continue
planes = int(shape[-1]) if len(shape) >= 3 else 1
return planes, name, len(npys)
return None, "", len(npys)
def _candidate_patterns(settings: Dict[str, Any]) -> List[Tuple[str, str]]:
"""``(label, pattern)`` regexes to try against raw filenames, best guess first."""
metadata_type = settings.get("metadata_type", "cellvoyager")
custom = settings.get("custom_regex")
out: List[Tuple[str, str]] = []
if isinstance(custom, str) and custom.strip():
out.append(("custom_regex", custom))
out.append(("custom_regex+ext", custom.rstrip("$") + _EXT_SUFFIX))
if isinstance(metadata_type, str) and metadata_type in METADATA_REGEXES:
out.append((metadata_type, METADATA_REGEXES[metadata_type]))
for label, pattern in METADATA_REGEXES.items():
if all(label != existing for existing, _ in out):
out.append((label, pattern))
return out
def _scan_raw_images(src: str, settings: Dict[str, Any], inv: _Inventory) -> None:
"""Fill ``inv`` with the channel/field counts implied by raw image filenames.
Channel identity comes from the ``chanID`` group of the metadata regex,
exactly as spacr.utils._extract_filename_metadata reads it; the stack
written by spacr.io._rename_and_organize_image_files concatenates one
plane per distinct chanID in sorted order, so the number of distinct
chanIDs *is* the number of valid zero-based channel indices.
"""
search_dirs = [src, os.path.join(src, "orig"), os.path.join(src, "consolidated")]
names: List[str] = []
for directory in search_dirs:
found = [f for f in _listdir(directory) if f.lower().endswith(IMAGE_EXTENSIONS)]
if found:
names = found
inv.raw_dir = directory
break
inv.raw_files = len(names)
if not names:
return
best_label = ""
best_channels: List[str] = []
best_fields = 0
best_hits = 0
for label, pattern in _candidate_patterns(settings):
try:
rx = re.compile(pattern)
except re.error:
continue
channels = set()
fields = set()
hits = 0
for name in names:
match = rx.match(name)
if not match:
continue
groups = match.groupdict()
if "chanID" not in groups or groups["chanID"] is None:
continue
hits += 1
channels.add(str(groups["chanID"]))
fields.add((
str(groups.get("plateID") or ""),
str(groups.get("wellID") or ""),
str(groups.get("fieldID") or ""),
str(groups.get("timeID") or ""),
))
if hits > best_hits:
best_hits, best_label = hits, label
best_channels = sorted(channels)
best_fields = len(fields)
if best_hits:
inv.raw_channels = len(best_channels)
inv.raw_channel_ids = tuple(best_channels)
inv.fields = best_fields or None
inv.regex_used = best_label
inv.raw_evidence = (
f"{best_hits} raw image files parsed with the '{best_label}' filename pattern"
)
def _inventory(src: Any, settings: Dict[str, Any], app: str) -> _Inventory:
"""Establish what is on disk for one ``src``, cheaply."""
inv = _Inventory()
if not isinstance(src, str):
return inv
if app in MERGED_APPS:
resolved, note = _resolve_measure_src(src)
if note:
inv.merged_note = note
src = os.path.dirname(resolved)
inv.src = src
inv.exists = os.path.exists(src)
inv.is_dir = os.path.isdir(src)
inv.is_db_file = inv.exists and not inv.is_dir and src.endswith(".db")
if not inv.is_dir:
return inv
inv.merged_dir = _resolve_merged_dir(src)
inv.merged_exists = os.path.isdir(inv.merged_dir)
if inv.merged_exists:
planes, example, count = _peek_planes(inv.merged_dir)
inv.merged_files = count
if planes is not None:
inv.array_planes = planes
inv.array_source = os.path.join(os.path.basename(inv.merged_dir), example)
inv.stack_dir = os.path.join(src, "stack")
if os.path.isdir(inv.stack_dir):
planes, example, count = _peek_planes(inv.stack_dir)
inv.stack_files = count
if planes is not None:
inv.raw_channels = planes
inv.raw_channel_ids = tuple(str(i) for i in range(planes))
inv.raw_evidence = f"stack/{example} has {planes} channel planes"
if inv.array_planes is None:
inv.array_planes = planes
inv.array_source = os.path.join("stack", example)
if inv.raw_channels is None:
_scan_raw_images(src, settings, inv)
db_root = os.path.dirname(os.path.normpath(src)) if os.path.basename(
os.path.normpath(src)).endswith("merged") else src
inv.db_path = os.path.join(db_root, "measurements", "measurements.db")
inv.db_exists = os.path.isfile(inv.db_path)
return inv
def _source_key(app: str) -> str:
"""Name of the setting holding this app's input folder — usually ``src``."""
return ALT_SRC_KEYS.get(app, "src")
def _src_values(settings: Dict[str, Any], app: str = "") -> List[Any]:
"""The app's source folder(s), mirroring spacr.utils.normalize_src_path."""
src = settings.get(_source_key(app))
if app == "external_masks":
values = src if isinstance(src, (list, tuple)) else [src]
roots: List[Any] = []
for value in values:
if isinstance(value, dict):
root = value.get("root")
if root:
roots.append(root)
continue
paths = value.get("paths") or []
roots.extend(
os.path.dirname(path) if os.path.isfile(path) else path
for path in paths if isinstance(path, str))
elif isinstance(value, str):
roots.append(
os.path.dirname(value) if os.path.isfile(value) else value)
return list(dict.fromkeys(roots))
if isinstance(src, (list, tuple)):
return list(src)
if isinstance(src, str):
stripped = src.strip()
if stripped.startswith("[") and stripped.endswith("]"):
import ast
try:
parsed = ast.literal_eval(stripped)
except (ValueError, SyntaxError):
return [src]
if isinstance(parsed, list):
return parsed
return [src]
return [src]
def _check_regression_output_src(raw: Any) -> List[Problem]:
"""Validate the optional regression output root without creating it.
Regression reads its score and count tables through dedicated settings;
``src`` names only the output root. A missing final directory is valid
when its parent exists because :func:`spacr.ml.resolve_regression_src`
creates that one directory. Configurations that trigger the documented
automatic fallback are warnings rather than errors because the analysis
remains runnable.
:param raw: Configured ``src`` value.
:returns: Problems associated with the output-root configuration.
"""
if raw is None or (isinstance(raw, str) and not raw.strip()):
return []
if not isinstance(raw, str):
return [Problem(
ERROR,
"src",
f"regression src={raw!r} is not a path string.",
"Set src to one output directory, or leave it blank to write "
"beside the first count table.",
)]
if raw in ("path", "/path/to/src"):
return [Problem(
ERROR,
"src",
f"regression src is still the placeholder {raw!r}.",
"Choose an output directory, or clear src to use the automatic "
"location beside the first count table.",
)]
requested = os.path.abspath(os.path.expanduser(raw.strip()))
if os.path.isdir(requested):
return []
if os.path.exists(requested):
return [Problem(
WARNING,
"src",
f"regression output path is not a directory: {requested}.",
"Choose a directory. If unchanged, spaCR will report the issue "
"and write beside the first count table instead.",
)]
parent = os.path.dirname(requested)
if os.path.isdir(parent):
return []
return [Problem(
WARNING,
"src",
f"regression output directory cannot be created because its parent "
f"does not exist: {parent}.",
"Create the parent directory or choose another output root. If "
"unchanged, spaCR will write beside the first count table instead.",
)]
def _check_src(settings: Dict[str, Any], app: str, inventories: Sequence[_Inventory]) -> List[Problem]:
"""``src`` exists, is the right kind of thing, and holds what the app needs."""
problems: List[Problem] = []
key = _source_key(app)
if key == "images":
fix = "Set images to the folder holding their images."
elif key == "inputs":
fix = (
"Drop intensity images and label masks onto External Masks, "
"then review their assignments.")
elif key == "genotype_source":
fix = ("Set genotype_source to the folder of sequencing tiles; OPS "
"searches it recursively.")
else:
fix = (
"Set src to the folder holding the images (or, for measure, "
"the merged folder).")
if app == "regression":
return _check_regression_output_src(settings.get(key))
if key not in settings:
return [Problem(ERROR, key, f"{key} is missing from the settings.", fix)]
raw = settings.get(key)
if raw is None or (isinstance(raw, str) and not raw.strip()):
return [Problem(ERROR, key, f"{key} is empty.", fix)]
for value in _src_values(settings, app):
if not isinstance(value, str):
problems.append(Problem(
ERROR, key,
f"{key} entry {value!r} is a {type(value).__name__}, not a path string.",
f"{key} must be a path string or a list of path strings."))
continue
elif value in ("path", "/path/to/src"):
problems.append(Problem(
ERROR, key,
f"{key} is still the placeholder {value!r} that the defaults ship with.",
"Replace it with the real folder you want to process."))
for inv in inventories:
if not isinstance(inv.src, str) or not inv.src:
continue
if not inv.exists:
problems.append(Problem(
ERROR, key, f"{key} does not exist: {inv.src}",
"Check the path for typos, and that the drive or share is mounted."))
continue
if inv.is_db_file:
continue
if not inv.is_dir:
problems.append(Problem(
ERROR, key, f"{key} is a file, not a folder: {inv.src}",
f"Point {key} at the folder that contains the images."))
continue
if app in MERGED_APPS:
if inv.merged_note:
problems.append(Problem(
WARNING, "src", inv.merged_note,
"Set src to the plate folder, or to its merged folder, to silence this."))
if not inv.merged_exists:
problems.append(Problem(
ERROR, "src",
f"no merged folder for measure: {inv.merged_dir} does not exist.",
"Run the Mask module on this plate first; measure reads the merged/*.npy it writes."))
elif inv.merged_files == 0:
problems.append(Problem(
ERROR, "src",
f"{inv.merged_dir} exists but contains no .npy arrays.",
"Re-run the Mask module: merged/ is written at the end of mask generation and is empty here."))
elif app in MASK_APPS:
if inv.raw_files == 0 and inv.stack_files == 0 and inv.merged_files == 0:
problems.append(Problem(
ERROR, "src",
f"no image files found in {inv.src} (looked for {', '.join(IMAGE_EXTENSIONS)}).",
"Point src at the folder that holds the raw acquisition "
"images. If they sit in per-well or per-channel "
"subfolders, run Import on this folder first — it reads "
"the subfolder names as part of the image name and writes "
"a plate the Mask module can read. consolidate=True is "
"the older path and copies the whole plate into "
"src/consolidated, which only helps when the flattened "
"names still match metadata_type."))
elif inv.raw_files and inv.raw_channels is None:
problems.append(Problem(
WARNING, "metadata_type",
f"{inv.raw_files} image files found in {inv.src}, but none match the "
f"'{settings.get('metadata_type', 'cellvoyager')}' filename pattern.",
"Set metadata_type to match your microscope — the dropdown now groups the built-in conventions by vendor, and 'Test on my folder' beside it reports how many of these files each one parses. Failing that, supply custom_regex with wellID/fieldID/chanID groups."))
if app in DB_APPS and not inv.db_exists:
problems.append(Problem(
ERROR, "src", f"measurements database not found: {inv.db_path}",
"Run the Measure module on this plate first — the analysis apps read measurements/measurements.db."))
return problems
def _as_int(value: Any) -> Optional[int]:
"""Return ``value`` as an int when it unambiguously is one, else None."""
if isinstance(value, bool):
return None
if isinstance(value, int):
return value
if isinstance(value, float) and float(value).is_integer():
return int(value)
if isinstance(value, str):
text = value.strip()
if text.lstrip("-").isdigit():
return int(text)
return None
def _check_channels(settings: Dict[str, Any], app: str, inventories: Sequence[_Inventory]) -> List[Problem]:
"""Channel indices are in range, and don't collide.
``*_channel`` indexes the raw acquisition channels; ``*_mask_dim`` indexes
the last axis of the merged array (image channels plus the appended mask
planes) — see spacr.io._load_and_concatenate_arrays.
"""
problems: List[Problem] = []
n_raw = next((inv.raw_channels for inv in inventories if inv.raw_channels is not None), None)
raw_evidence = next((inv.raw_evidence for inv in inventories if inv.raw_channels is not None), "")
n_planes = next((inv.array_planes for inv in inventories if inv.array_planes is not None), None)
plane_evidence = next((inv.array_source for inv in inventories if inv.array_planes is not None), "")
for key in CHANNEL_KEYS:
if key not in settings:
continue
value = settings[key]
if value is None:
continue
index = _as_int(value)
if index is None:
continue
if index < 0:
problems.append(Problem(
ERROR, key, f"{key}={value} is negative.",
f"Channel indices are zero-based; use 0"
+ (f"-{n_raw - 1}" if n_raw else "") + ", or None to skip this object."))
continue
if n_raw is not None and index >= n_raw:
problems.append(Problem(
ERROR, key,
f"{key}={index} but the dataset has only {n_raw} channel"
f"{'' if n_raw == 1 else 's'} ({raw_evidence}); valid indices are "
f"0-{n_raw - 1}.",
f"Set {key} to a value between 0 and {n_raw - 1}, or to None if that stain was not acquired."))
elif n_raw is None and n_planes is not None and index >= n_planes:
problems.append(Problem(
ERROR, key,
f"{key}={index} is past the end of the stored arrays, which have "
f"{n_planes} planes ({plane_evidence}).",
f"Set {key} to at most {n_planes - 1}, or to None if that stain was not acquired."))
for key in MASK_DIM_KEYS:
if key not in settings:
continue
value = settings[key]
if value is None:
continue
index = _as_int(value)
if index is None:
continue
if index < 0:
problems.append(Problem(
ERROR, key, f"{key}={value} is negative.",
"Mask dims are zero-based positions along the last axis of merged/*.npy; use None to skip the object."))
continue
if n_planes is not None and index >= n_planes:
problems.append(Problem(
ERROR, key,
f"{key}={index} is past the end of the merged arrays, which have "
f"{n_planes} planes ({plane_evidence}); valid positions are 0-{n_planes - 1}.",
f"With {n_planes} planes the masks sit at the top of the stack. "
f"Set {key} within 0-{n_planes - 1}, or None to skip that object."))
channels = settings.get("channels")
if isinstance(channels, (list, tuple)) and n_planes is not None:
bad = [c for c in channels if (_as_int(c) is not None and _as_int(c) >= n_planes)]
if bad:
problems.append(Problem(
ERROR, "channels",
f"channels {bad} are past the end of the stored arrays, which have "
f"{n_planes} planes ({plane_evidence}).",
f"Reduce channels to indices within 0-{n_planes - 1}."))
problems.extend(_collision_problems(settings, CHANNEL_KEYS, WARNING,
"are both assigned channel",
"Give each object its own acquisition channel, unless you really mean to segment both from the same stain."))
problems.extend(_collision_problems(settings, MASK_DIM_KEYS, ERROR,
"both read mask plane",
"Each object needs its own mask plane; with four image channels the masks land at 4, 5, 6, 7 in the order cell, nucleus, pathogen, organelle."))
return problems
def _collision_problems(settings: Dict[str, Any], keys: Sequence[str],
severity: str, phrase: str, fix: str) -> List[Problem]:
"""Report keys in ``keys`` that share an index value."""
by_value: Dict[int, List[str]] = {}
for key in keys:
index = _as_int(settings.get(key))
if index is None or index < 0:
continue
by_value.setdefault(index, []).append(key)
out: List[Problem] = []
for index, sharers in sorted(by_value.items()):
if len(sharers) > 1:
out.append(Problem(
severity, ", ".join(sharers),
f"{' and '.join(sharers)} {phrase} {index}.", fix))
return out
def _type_name(expected: Any) -> str:
"""Render an ``expected_types`` entry as readable prose."""
if isinstance(expected, tuple):
return " or ".join(
"None" if t is type(None) else getattr(t, "__name__", str(t))
for t in expected)
return getattr(expected, "__name__", str(expected))
_EXPECTED_TYPE_OVERRIDES: Dict[str, Any] = {
"src": (str, list),
"normalize": (bool, list),
"save": (bool, list),
}
_APP_TYPE_OVERRIDES: Dict[str, Dict[str, Any]] = {
"foreign": {"masks": (str, list)},
}
[docs]
def coerce_expected_types(settings: Dict[str, Any],
app: str = "") -> Dict[str, Any]:
"""Return ``settings`` with text-written numbers as their declared type.
A settings CSV round-trip makes every value a string, and so does a number
typed into a GUI field. ``expected_types`` is the contract those values are
meant to satisfy, so converting them to it is restoring what the settings
already claim to be -- not reinterpreting them.
Doing it HERE, once, at the boundary, rather than at each point of use, is
what stops the next consumer from being the one that crashes: mask
generation died inside Cellpose on ``diameter > 0`` with
``cell_diameter='60.0'``, and the same file had already been reported,
three times, as an error the user was told to fix by hand -- for a value
that was perfectly well-formed.
CONSERVATIVE BY CONSTRUCTION. Only ``bool``, ``int`` and ``float`` are
converted, only from a string, only when the conversion is exact, and
never when ``str`` is itself an accepted type for the key. Anything that
does not convert cleanly is left exactly as it was, for
:func:`validate_settings` to report.
:param settings: the settings mapping.
:param app: the pipeline, for the same per-app type overrides
:func:`_check_types` honours.
:returns: a new dict; the input is not modified.
"""
from .settings import expected_types
per_app = _APP_TYPE_OVERRIDES.get(app, {})
out = dict(settings)
for key, value in settings.items():
if not isinstance(value, str) or key not in expected_types:
continue
text = value.strip()
if not text:
continue
expected = per_app.get(
key, _EXPECTED_TYPE_OVERRIDES.get(key, expected_types[key]))
types = expected if isinstance(expected, tuple) else (expected,)
if str in types:
continue
if bool in types:
lowered = text.lower()
if lowered in ("true", "yes", "1"):
out[key] = True
elif lowered in ("false", "no", "0"):
out[key] = False
continue
try:
number = float(text)
except ValueError:
continue
if int in types and float not in types:
if number.is_integer():
out[key] = int(number)
elif float in types:
out[key] = number
return out
def _check_types(settings: Dict[str, Any], app: str = "") -> List[Problem]:
"""Values match ``spacr.settings.expected_types``.
A settings CSV round-trip is the usual culprit: every value comes back a
string, so ``cell_mask_dim`` arrives as ``'4'`` and measure_crop bails out
with "must all be integers".
"""
from .settings import expected_types
per_app = _APP_TYPE_OVERRIDES.get(app, {})
problems: List[Problem] = []
for key, value in settings.items():
if key not in expected_types:
continue
expected = per_app.get(
key, _EXPECTED_TYPE_OVERRIDES.get(key, expected_types[key]))
types = expected if isinstance(expected, tuple) else (expected,)
if value is None:
continue
if isinstance(value, tuple) and list in types:
continue
if isinstance(value, bool) and bool not in types:
problems.append(Problem(
WARNING, key,
f"{key}={value} is a bool but is declared as {_type_name(expected)}.",
f"Set {key} to a {_type_name(expected)} value."))
continue
if isinstance(value, int) and not isinstance(value, bool) and float in types and int not in types:
continue
if not isinstance(value, types):
hint = ""
if isinstance(value, str):
hint = (" Values read back from a settings CSV are strings; "
"re-import through the settings panel or convert it by hand.")
problems.append(Problem(
ERROR, key,
f"{key}={value!r} is a {type(value).__name__}, but "
f"{_type_name(expected)} is expected.",
f"Set {key} to a {_type_name(expected)}.{hint}"))
return problems
_APP_EXTRA_KEYS: Dict[str, frozenset] = {
"measure": frozenset({"_psf_measurement_signature"}),
"foreign": frozenset({
"images", "masks", "measurements", "dst", "layout", "z_handling",
"plate_naming", "measurement_table", "measurement_object", "image_key",
"label_key", "column_map", "um_per_px", "on_conflict",
"allow_spacr_targets", "measure", "crops", "overwrite", "preview_only",
}),
"external_masks": frozenset({
"inputs", "dst", "recursive", "layout", "z_handling",
"plate_naming", "overwrite", "preview_only",
}),
}
#: Settings that spaCR used to offer, and what replaced each.
#:
#: A RETIRED KEY IS THE ONE CASE FUZZY MATCHING CANNOT REACH. The check
#: below only speaks up when a live setting is within a close match of the
#: name it was handed, and that is deliberate: newer pipelines carry keys
#: this module has never heard of, so warning about every one of them would
#: be constant noise. But when a setting is deleted, its nearest neighbours
#: usually go with it -- `upscale` and `upscale_factor` were removed in the
#: same breath -- so nothing is left within matching distance and an old
#: settings file naming one gets no warning at all. The value is ignored,
#: the default is used, and the run differs from the file that describes it
#: without a word.
#:
#: That is the worst place for silence, because a retired name is exactly
#: what an OLD settings file contains, and its author has every reason to
#: believe it still applies.
#:
#: An empty string means the setting was removed outright rather than
#: renamed. Only renames that were verified against the live settings are
#: recorded as such; a guess here would send a user to a name that is also
#: not read.
#: A value is the replacement's name, a TUPLE of names when the setting
#: was split in two, or an empty string when there is no replacement.
RETIRED_SETTINGS: Dict[str, Union[str, Tuple[str, ...]]] = {
"gradient_accumulation": "gradient_accumulation_steps",
"expected_end": "window_length",
"min_n": "min_observations_per_hit",
"min_cell_count": "min_cells_per_well",
"positive_control": "positive_control_id",
"negative_control": "negative_control_id",
"controls": "nontargeting_control_grnas",
"control_wells": ("stain_baseline_wells", "analysis_excluded_wells"),
"denoise": "",
"load_path_regex": "",
"mask_array": "",
"normalization": "",
"normalization_scope": "",
"normalize_plots": "",
"save_to_db": "",
"visualize": "",
"organelle_min_size": "organelle_min_area",
"organelle_max_size": "organelle_max_area",
"minimum_cell_count": "min_cells_per_well",
"redunction_method": "reduction_method",
"barcode_coordinates": "",
"barcode_mapping": "",
"compartments": "",
"compression": "",
"complevel": "comp_level",
"correlate": "",
"downstream": "",
"upstream": "",
"split_axis_lims": "",
"upscale": "",
"upscale_factor": "",
"all_to_mip": "",
"custom_measurement": "",
"gene_weights_csv": "",
"metadata_types": "",
"pick_slice": "",
"skip_mode": "",
"signal_direction": "",
"measurement_rules": "",
"cells_per_page": "",
"extract_channels": "",
"infection_xgb_proba": "",
"highlight": "",
"guide_permutation_plot": "",
"corrected_manders": "",
"grna": "",
"barcodes": "",
"Toxoplasma": "annotation_source",
"toxo": "annotation_source",
"img_size": "crop_size",
"infection_pca_n_clusters": "",
"straightness_filter": "drop_straight_tracks",
"zscore_thresh": "track_outlier_zscore",
**{f"{obj}_{bound}": "object_filters"
for obj in ("cell", "nucleus", "pathogen")
for bound in ("min_area", "max_area", "min_intensity",
"max_intensity")},
}
#: NOT HERE: a setting withdrawn from ONE panel while `spacr.settings` still
#: declares it. `log_x`, `log_y`, `x_lim`, `y_lims` and `png_type` left the
#: regression panel and are read elsewhere, so naming one here would warn a
#: user off a setting that works.
def _check_retired_keys(settings: Dict[str, Any]) -> List[Problem]:
"""Say so when a settings file names a setting spaCR has withdrawn.
Separate from the typo check because the two have opposite shapes: a
typo is caught by resembling something real, and a retired name is
missed for the same reason -- whatever it resembled was withdrawn with
it.
"""
problems: List[Problem] = []
for key in settings:
if not isinstance(key, str):
continue
replacement = RETIRED_SETTINGS.get(key)
if replacement is None:
from .settings import surviving_setting_name
survivors = surviving_setting_name(key)
if not survivors:
from .object_roles import (split_role_setting,
withdrawn_setting_reason)
gone = withdrawn_setting_reason(key)
if gone:
problems.append(Problem(
WARNING, key,
f"'{key}' is no longer a spaCR setting.",
f"Remove '{key}' — {gone}. As it stands the value is "
f"read by nothing."))
continue
replacement = (survivors[0] if len(survivors) == 1
else tuple(survivors))
from .settings import SEMANTIC_FOLD_MEANINGS
meaning = SEMANTIC_FOLD_MEANINGS.get(key)
if meaning and replacement:
problems.append(Problem(
WARNING, key,
f"'{key}' was folded into '{replacement}'.",
f"Set '{replacement}' instead. spaCR still reads the old "
f"value -- {meaning} -- so a file that has not been updated "
f"still behaves as it did."))
continue
if isinstance(replacement, (tuple, list)):
names = ", ".join(f"'{one}'" for one in replacement)
problems.append(Problem(
WARNING, key,
f"'{key}' was split into {names}.",
f"Set whichever of {names} you meant — spaCR applies the "
f"old value to both, so a file that has not been updated "
f"still behaves as it did."))
continue
if replacement:
problems.append(Problem(
WARNING, key,
f"'{key}' was renamed to '{replacement}'.",
f"Rename '{key}' to '{replacement}'. spaCR still moves the "
f"value across when it reads this file, but the new name is "
f"the one to write."))
else:
problems.append(Problem(
WARNING, key,
f"'{key}' is no longer a spaCR setting.",
f"Remove '{key}' — spaCR does not read it, so the value "
f"has no effect on the run."))
return problems
def _object_role_in(key):
"""The object role a settings key names, at EITHER end, or ``None``.
`<role>_suffix` is the common shape -- `cell_min_area` -- and
`object_roles.split_role_setting` handles it. But the shape that actually
produced a wrong suggestion puts the role LAST: `remove_background_organelle`
was matched to `remove_background_cell`, because the only difference is the
final word and `difflib` scores on characters. A guard that looked only at
the front would have missed the one case anybody has hit.
An organelle slot's background switch names its slot by number at the
end, ``remove_background_organelle_7``; that is slot 7's role, so slot 1's
switch is never answered with slot 7's.
:param key: a settings key.
:returns: the role name, or ``None`` when the key names none.
"""
from .object_roles import ALL_ROLES, split_role_setting
from .organelle_types import _background_switch_role
switch = _background_switch_role(key)
if switch is not None:
return switch
parts = split_role_setting(key)
if parts is not None:
return parts[0]
tail = str(key).rsplit("_", 1)[-1]
return tail if tail in ALL_ROLES else None
_MODULE_DEFAULT_KEYS_CACHE: Dict[str, frozenset] = {}
def _module_default_keys(app: str) -> frozenset:
"""Every key the defaults of the headless modules validated as ``app`` produce.
A module that keeps its own defaults (convert, illumination, ...) owns
settings no :mod:`spacr.settings` helper lists, and a settings file built
from those very defaults must not be told its keys are unknown. Read
through :func:`spacr.cli.module_defaults`, so the answer is the dict the
run itself starts from; a module whose defaults will not import adds
nothing.
:param app: canonical app key, as :func:`validate_settings` uses it.
:returns: the union of those modules' default keys; empty for no app.
"""
if not app:
return frozenset()
cached = _MODULE_DEFAULT_KEYS_CACHE.get(app)
if cached is not None:
return cached
keys: set = set()
try:
from .cli import MODULES, module_defaults
except Exception:
return frozenset()
for module in MODULES.values():
if app not in (module.key, module.validate_key):
continue
try:
produced = module_defaults(module)
except Exception:
continue
keys.update(k for k in produced if isinstance(k, str))
_MODULE_DEFAULT_KEYS_CACHE[app] = frozenset(keys)
return _MODULE_DEFAULT_KEYS_CACHE[app]
def _check_unknown_keys(settings: Dict[str, Any], app: str = "") -> List[Problem]:
"""Flag keys spaCR does not know: a likely typo, or simply unknown.
Decision 2026-09-25 (item 237): "settings files with retired keys: WARN
AND MIGRATE -- known retired keys are migrated to their successors;
truly unknown keys produce a visible warning but the run continues."
A key with a close match to a live setting is reported as a typo with
the suggestion. A key with no close match, that is not retired
(:func:`_check_retired_keys` speaks for those) and not renamed, is now
reported too -- as a WARNING, never an ERROR, so the run goes ahead
with the value ignored. "Known" is broad: ``expected_types``, the
tooltips, the category lists and every key a ``set_default_*`` /
``get_*_settings`` helper produces, plus the app's own extra keys and
the default keys of the module itself (:func:`_module_default_keys`). A
plugin app's settings are its own, so for one only the typo check runs.
A key with no value is not reported: older settings files carry their
section headings ("General", "Cell", ...) as blank rows, and a blank
value changes nothing whatever its name.
"""
known = (_known_setting_keys() | _APP_EXTRA_KEYS.get(app, frozenset())
| _module_default_keys(app))
try:
from .plugins import get_app as _get_plugin_app
plugin = bool(app) and _get_plugin_app(app) is not None
except Exception:
plugin = False
problems: List[Problem] = []
for key in settings:
if not isinstance(key, str) or key in known:
continue
if key in RETIRED_SETTINGS:
continue
from .object_roles import withdrawn_setting_reason
from .settings import surviving_setting_name
if surviving_setting_name(key) or withdrawn_setting_reason(key):
continue
close = difflib.get_close_matches(key, sorted(known), n=5, cutoff=0.85)
mine = _object_role_in(key)
if mine is not None:
close = [c for c in close if _object_role_in(c) in (None, mine)]
if close:
problems.append(Problem(
WARNING, key,
f"'{key}' is not a spaCR setting; did you mean '{close[0]}'?",
f"Rename '{key}' to '{close[0]}' — as it stands the value is ignored and the default is used."))
elif (not plugin and not key.startswith("_")
and settings[key] not in (None, "")):
problems.append(Problem(
WARNING, key,
f"'{key}' is not a setting spaCR knows.",
f"The run continues and '{key}' is ignored. Remove it from "
f"the settings file, or check the spelling if it was meant "
f"to change something."))
return problems
def _numeric(value: Any) -> Optional[float]:
"""Return ``value`` as a float when it is a real number, else None."""
if isinstance(value, bool):
return None
if isinstance(value, (int, float)):
return float(value)
return None
_FLOW_THRESHOLD_DEFAULT = 0.4
def _flow_threshold_problems(key: str, value: Any, number: float) -> List[Problem]:
"""Warn when a Cellpose flow threshold leaves the filter doing nothing.
Cellpose discards a mask whose flow error -- the mean squared difference
between the flows recomputed from the mask and the flows the network
predicted, both of about unit length -- is above the threshold. Above 3
that rejects practically nothing, and at 0 or below Cellpose skips the
check, so either way every mask Cellpose proposes is kept. Values above
3 and below 0 are reported. Exactly 0 is not: it is Cellpose's own
switch for turning the check off ("turn off QC step with
flow_threshold=0 if too slow"), so it is taken as meant. The shipped
default of 0.4 is never reported, and the 100 that spaCR 1.5.0.5 to
1.5.0.8 shipped still is.
:param key: the setting name.
:param value: the value as the settings hold it.
:param number: ``value`` as a float.
:returns: one warning when the value is above 3 or below 0, else
nothing.
"""
if number > 3:
return [Problem(
WARNING, key,
f"{key}={value} turns Cellpose's flow-error filter off: above 3 "
"it rejects practically nothing, so every mask Cellpose proposes "
"is kept, however misshapen.",
f"spaCR and Cellpose both default to {_FLOW_THRESHOLD_DEFAULT}. "
"spaCR 1.5.0.5 to 1.5.0.8, and some older releases, shipped 100, "
"so a settings file saved by one of them still carries it. Set "
f"{key} to {_FLOW_THRESHOLD_DEFAULT} to drop misshapen masks again "
"(values above 0 and up to 3 filter, and lower keeps fewer, "
"cleaner objects), or keep it above 3 only if you want every "
"candidate.")]
if number < 0:
return [Problem(
WARNING, key,
f"{key}={value} is below 0, and Cellpose skips its flow-error "
"filter for any value of 0 or less, so every mask Cellpose "
"proposes is kept.",
f"spaCR and Cellpose both default to {_FLOW_THRESHOLD_DEFAULT}. "
f"Set {key} above 0 and at most 3 to filter misshapen masks; "
"lower keeps fewer, cleaner objects.")]
return []
def _check_numeric_sanity(settings: Dict[str, Any]) -> List[Problem]:
"""Diameters, percentiles, thresholds and batch sizes are in usable ranges."""
problems: List[Problem] = []
for key, value in settings.items():
if not isinstance(key, str):
continue
number = _numeric(value)
if number is not None and (key.endswith("_diameter") or key == "diameter"):
if number <= 0:
problems.append(Problem(
ERROR, key, f"{key}={value} must be greater than zero.",
"Give the expected object size in pixels, or None to let magnification derive it."))
if number is not None and key in (
"batch_size", "test_images", "test_nr", "nr_imgs",
"epochs", "n_epochs", "image_size", "size", "chunk_size",
"magnification", "examples_to_plot", "guide_permutations"):
if number < 1:
problems.append(Problem(
ERROR, key, f"{key}={value} must be at least 1.",
f"Set {key} to a positive whole number."))
if number is not None and key == "n_jobs":
if number < 1 and number != -1:
problems.append(Problem(
ERROR, key, f"n_jobs={value} is not a usable worker count.",
"Use a positive number of CPU workers, -1 for every core, or leave it blank."))
if number is not None and (key.endswith("_percentile") or key in ("lower_percentile", "upper_percentile")):
if not 0 <= number <= 100:
problems.append(Problem(
ERROR, key, f"{key}={value} is not a percentile (0-100).",
f"Set {key} between 0 and 100."))
if number is not None and (key.endswith("_cellprob_threshold") or key in ("CP_prob", "CP_probability")):
if not -6 <= number <= 6:
problems.append(Problem(
WARNING, key, f"{key}={value} is outside Cellpose's usable -6 to 6 range.",
"Lower it toward -6 to grow masks and keep faint objects; raise it toward 6 to shrink them."))
if number is not None and (key.endswith("_flow_threshold") or key in ("FT", "flow_threshold")):
problems.extend(_flow_threshold_problems(key, value, number))
if number is not None and key in ("val_split", "test_split", "dropout_rate",
"organelle_unet_threshold", "score_threshold"):
if not 0 <= number <= 1:
problems.append(Problem(
ERROR, key, f"{key}={value} must be a fraction between 0 and 1.",
f"Set {key} between 0 and 1 (0.1 means 10%)."))
if number is not None and key == "learning_rate" and number <= 0:
problems.append(Problem(
ERROR, key, f"learning_rate={value} must be greater than zero.",
"Typical values are 1e-4 to 1e-2."))
if isinstance(value, (list, tuple)):
problems.extend(_check_numeric_list(key, value))
return problems
def _check_numeric_list(key: str, value: Sequence[Any]) -> List[Problem]:
"""Range-check list-valued percentile settings."""
problems: List[Problem] = []
percentile_lists = ("normalization_percentiles", "manders_thresholds",
"percentiles", "normalize")
if key in percentile_lists or key.endswith("_percentiles"):
numbers = [_numeric(v) for v in value]
if any(n is None for n in numbers) or not numbers:
return problems
out_of_range = [n for n in numbers if not 0 <= n <= 100]
if out_of_range:
problems.append(Problem(
ERROR, key, f"{key}={list(value)} contains values outside 0-100.",
f"{key} is a list of percentiles; every entry must be between 0 and 100."))
elif len(numbers) == 2 and numbers[0] >= numbers[1]:
problems.append(Problem(
ERROR, key,
f"{key}={list(value)} has its lower percentile at or above its upper percentile.",
f"Write {key} as [lower, upper], for example [1, 99]."))
if key == "png_size":
flat = value if value and not isinstance(value[0], (list, tuple)) else [
v for pair in value for v in pair]
numbers = [_numeric(v) for v in flat]
if numbers and all(n is not None for n in numbers) and any(n < 1 for n in numbers):
problems.append(Problem(
ERROR, key, f"png_size={list(value)} contains a non-positive size.",
"png_size is [width, height] in pixels, for example [224, 224]."))
return problems
def _check_required_paths(settings: Dict[str, Any], app: str) -> List[Problem]:
"""App-specific inputs that must already exist on disk."""
problems: List[Problem] = []
def _require_file(key: str, purpose: str, fix: str) -> None:
"""Refuse a missing or unset path, saying what it was needed FOR.
Both the purpose and the fix are carried into the message: "not set" on
its own tells a user what happened and not what to do about it.
"""
value = settings.get(key)
if value is None or (isinstance(value, str) and not value.strip()):
problems.append(Problem(
ERROR, key, f"{key} is not set, but {purpose}.", fix))
return
if isinstance(value, str) and not os.path.isfile(value):
if key == "model_path":
from .model_zoo import ModelZooError, _ensure_model_file
try:
if _ensure_model_file(
value, kinds=("classifier",), download=False) is not None:
return
except ModelZooError:
pass
problems.append(Problem(
ERROR, key, f"{key} points at a file that does not exist: {value}", fix))
if app == "map_barcodes":
for key, label in (("grna_csv", "the gRNA barcodes"),
("row_csv", "the row barcodes"),
("column_csv", "the column barcodes")):
_require_file(
key, f"barcode mapping needs {label}",
f"Point {key} at a CSV with 'name' and 'sequence' columns.")
if app == "foreign":
for key, purpose in (("images", "the import reads their images"),
("masks", "the import reads their mask folder(s)"),
("measurements", "the import reads their measurement table")):
value = settings.get(key)
if value is None or (isinstance(value, str) and not value.strip()) or value == []:
problems.append(Problem(
ERROR, key, f"{key} is not set, but {purpose}.",
f"Point {key} at what the other lab gave you."))
elif isinstance(value, str) and not os.path.exists(value):
problems.append(Problem(
ERROR, key, f"{key} points at a path that does not exist: {value}",
"Check the path, and that the share holding it is mounted."))
if not settings.get("column_map") and not settings.get("preview_only"):
problems.append(Problem(
WARNING, "column_map",
"no reviewed column_map, so the import will run with the columns "
"it inferred.",
"Run once with preview_only=True, save the column map, read it, "
"then point column_map at it."))
if app == "external_masks":
inputs = settings.get("inputs")
if not inputs:
problems.append(Problem(
ERROR, "inputs",
"No external images or masks have been selected.",
"Drop intensity images and label masks onto the module, then "
"review their assignments."))
else:
values = inputs if isinstance(inputs, (list, tuple)) else [inputs]
roles = {
value.get("role")
for value in values if isinstance(value, dict)
}
if roles and "image" not in roles:
problems.append(Problem(
ERROR, "inputs", "No input group is assigned as images.",
"Set at least one detected group to Image."))
if roles and "mask" not in roles:
problems.append(Problem(
ERROR, "inputs", "No input group is assigned as masks.",
"Set at least one detected group to Mask and choose its "
"object type."))
unassigned = [
value for value in values
if isinstance(value, dict)
and value.get("role") == "mask"
and value.get("object_type") not in OBJECT_NAMES
]
if unassigned:
problems.append(Problem(
ERROR, "inputs",
f"{len(unassigned)} mask group(s) have no object type.",
"Choose Cell, Nucleus, Pathogen or Organelle for every "
"mask group."))
if app in ("classify", "classify_merged"):
train = settings.get("train", settings.get("generate_training_dataset", False))
needs_model = bool(settings.get("apply_model_to_dataset", False)) or bool(settings.get("test", False))
if needs_model and not train:
_require_file(
"model_path", "scoring a dataset needs a trained classifier",
"Point model_path at a saved spaCR model, or set train=True to train one first.")
custom_model = settings.get("custom_model")
if isinstance(custom_model, str) and custom_model.strip():
if not os.path.exists(custom_model):
from .model_zoo import ModelZooError, _ensure_model_file
try:
downloadable = _ensure_model_file(
custom_model, kinds=("cellpose",), download=False) is not None
except ModelZooError:
downloadable = False
if not downloadable:
problems.append(Problem(
ERROR, "custom_model",
f"custom_model points at a path that does not exist: {custom_model}",
"Point custom_model at the saved Cellpose model file, or clear it to use the stock model."))
for organelle_role in SEGMENTED_ROLES[3:]:
method_key = f"{organelle_role}_method"
path_key = f"{organelle_role}_unet_model_path"
unet_path = settings.get(path_key)
if settings.get(method_key) != "unet":
continue
if not isinstance(unet_path, str) or not unet_path.strip():
problems.append(Problem(
ERROR, path_key,
f"{method_key}='unet' but no {path_key} is set.",
f"Point {path_key} at the serialised U-Net, or pick another {method_key}."))
elif not os.path.exists(unet_path):
problems.append(Problem(
ERROR, path_key,
f"{organelle_role} U-Net model not found: {unet_path}",
f"Fix the path, or pick another {method_key}."))
return problems
def _check_app_specific(settings: Dict[str, Any], app: str) -> List[Problem]:
"""Cross-setting rules the pipeline entry points enforce at runtime."""
problems: List[Problem] = []
if app == 'measure':
from .psf_measurement import prepare_measurement_psf
try:
prepare_measurement_psf(settings)
except (ValueError, OSError) as exc:
problems.append(Problem(
ERROR, 'psf_measurement_source', str(exc),
'Choose original intensities or configure a calibrated PSF for processed measurements.'))
if app in ('mask', 'timelapse') and settings.get('psf_operation', 'none') != 'none':
from .point_spread import fill_psf_settings
from .psf_pipeline import prepare_psf
try:
candidate = dict(settings)
fill_psf_settings(candidate)
prepare_psf(candidate)
except (ValueError, OSError) as exc:
problems.append(Problem(
ERROR, 'psf_operation', f'PSF preparation failed: {exc}',
'Set calibrated Y/X sampling and a matching measured kernel '
'or explicit Gaussian FWHM, or switch psf_operation to none.'))
if app in ('mask', 'timelapse'):
from .psf_pipeline import chain_problems
for key, message in chain_problems(settings):
problems.append(Problem(
ERROR, key, f'Image enhancement cannot run: {message}.',
'Choose one of the listed methods or a value in range, or '
'switch that enhancement step off.'))
if app == "explain_cv":
for key, label in (("db_path", "measurements database"),
("predictions_file", "prediction CSV")):
value = str(settings.get(key) or "").strip()
if not value:
problems.append(Problem(
ERROR, key, f"no {label} is selected.",
f"Set {key} to the exact existing file used by this run."))
elif not os.path.isfile(os.path.expanduser(value)):
problems.append(Problem(
ERROR, key, f"{label} does not exist: {value}",
f"Fix {key}; Explain CV Model never invents or reruns this input."))
family = str(settings.get("surrogate_model", "random_forest")).lower()
if family not in {"random_forest", "hist_gradient_boosting", "xgboost"}:
problems.append(Problem(
ERROR, "surrogate_model", f"unsupported surrogate family {family!r}.",
"Choose random_forest, hist_gradient_boosting, or xgboost."))
split = str(settings.get("surrogate_split_by", "well")).lower()
if split not in {"well", "plate"}:
problems.append(Problem(
ERROR, "surrogate_split_by", f"unsupported split unit {split!r}.",
"Choose well or plate; individual cells are not independent splits."))
if app == "investigate_hit":
for key in ("db_path", "predictions_file", "guide_fractions_file"):
value = str(settings.get(key) or "").strip()
if not value or not os.path.isfile(os.path.expanduser(value)):
problems.append(Problem(
ERROR, key, f"{key} is not an existing file: {value or '(blank)'}.",
f"Select the exact {key}; the investigation does not infer newest files."))
folder = str(settings.get("results_folder") or "").strip()
if not folder or not os.path.isdir(os.path.expanduser(folder)):
problems.append(Problem(
ERROR, "results_folder",
f"results_folder is not an existing regression folder: {folder or '(blank)' }.",
"Select the exact source run so its result bytes can be hashed."))
if not settings.get("target_gene") or not settings.get("target_guides"):
problems.append(Problem(
ERROR, "target_guides", "target_gene and target_guides are required.",
"Carry the selected hit and its exact supporting guide IDs from Hit List."))
if str(settings.get("hit_direction", "positive")) not in {"positive", "negative"}:
problems.append(Problem(
ERROR, "hit_direction", "hit_direction must be positive or negative.",
"Use the sign of the selected regression effect."))
if app in MASK_APPS:
if all(settings.get(k) is None for k in CHANNEL_KEYS):
problems.append(Problem(
ERROR, "cell_channel",
"no segmentation channel is set: every registered object channel is None.",
"Set at least one *_channel setting to an acquisition-channel index."))
if settings.get("pathogen_channel") is not None:
model = settings.get("pathogen_model")
from .settings import CELLPOSE_MODEL_CHOICES
stock_model = CELLPOSE_MODEL_CHOICES[0]
if isinstance(model, str) and model.strip():
model = model.strip()
looks_like_a_path = (
os.sep in model or model.endswith((".pth", ".pt")))
if looks_like_a_path and not os.path.isfile(model):
problems.append(Problem(
ERROR, "pathogen_model",
f"pathogen_model={model!r} names a checkpoint that "
f"is not there.",
"Point it at an existing .pth/.pt file, or drop the "
"setting to segment pathogens with stock cpsam."))
elif not looks_like_a_path and model != stock_model:
problems.append(Problem(
WARNING, "pathogen_model",
f"pathogen_model={model!r} is not a checkpoint on "
f"disk, so Cellpose 4 will load stock "
f"{stock_model!r} instead. The pre-SAM "
f"toxo_pv_lumen / toxo_cyto checkpoints are gone.",
"Give the path to a fine-tuned checkpoint, or set "
f"{stock_model!r} to be explicit."))
if app == "measure":
normalize = settings.get("normalize")
if isinstance(normalize, bool) and normalize:
problems.append(Problem(
ERROR, "normalize",
"normalize=True is rejected by measure_crop, which needs a percentile pair.",
"Use a two-element list such as [1, 99], or False to skip normalization."))
if isinstance(normalize, (list, tuple)) or normalize is True:
if settings.get("normalize_by") not in ("png", "fov"):
problems.append(Problem(
ERROR, "normalize_by",
f"normalize_by={settings.get('normalize_by')!r} is not understood.",
"Use 'png' to normalize each crop to its own percentiles, or 'fov' to use the whole field."))
crop_mode = settings.get("crop_mode")
if isinstance(crop_mode, (list, tuple)):
allowed = set(ALL_ROLES)
bad = [m for m in crop_mode if m not in allowed]
if bad:
problems.append(Problem(
ERROR, "crop_mode", f"crop_mode contains unsupported entries: {bad}.",
f"crop_mode entries must come from {sorted(allowed)}."))
ratios = settings.get("dialate_png_ratios")
if settings.get("dialate_pngs") and isinstance(ratios, (list, tuple)):
needed = len([m for m in crop_mode if m != "cytoplasm"])
if 1 < len(ratios) < needed:
problems.append(Problem(
WARNING, "dialate_png_ratios",
f"dialate_png_ratios has {len(ratios)} entries but "
f"{needed} crop modes are indexed against it; the last "
f"value will be reused for the rest.",
f"Give dialate_png_ratios one value (broadcast to every "
f"mode) or one per crop mode, e.g. {[0.2] * needed}."))
elif len(ratios) > needed:
problems.append(Problem(
WARNING, "dialate_png_ratios",
f"dialate_png_ratios has {len(ratios)} entries but only "
f"{needed} crop mode(s) read it; the extras are ignored.",
f"Trim dialate_png_ratios to {needed} value(s), or to a "
f"single value broadcast to every mode."))
for mode in crop_mode:
key = f"{mode}_mask_dim"
if mode in OBJECT_NAMES and settings.get(key) is None and key in settings:
problems.append(Problem(
ERROR, key,
f"crop_mode asks for {mode} crops but {key} is None, so no {mode} mask is read.",
f"Set {key} to the plane holding the {mode} mask, or drop '{mode}' from crop_mode."))
if app == "replication":
maximum = _as_int(settings.get("max_parasites_per_vacuole"))
if maximum is not None and (
maximum < 1 or maximum & (maximum - 1)):
problems.append(Problem(
ERROR, "max_parasites_per_vacuole",
f"max_parasites_per_vacuole={maximum} is not a positive "
"power of two.",
"Use 1, 2, 4, 8, 16, ... so every named bucket follows the "
"endodyogeny doubling ladder."))
factor = _numeric(settings.get("vacuole_link_factor"))
if factor is not None and factor <= 0:
problems.append(Problem(
ERROR, "vacuole_link_factor",
f"vacuole_link_factor={factor:g} cannot derive a positive "
"spatial linking distance.",
"Use a positive multiplier such as 1.5, or set "
"vacuole_link_distance explicitly."))
distance = _numeric(settings.get("vacuole_link_distance"))
if distance is not None and distance <= 0:
problems.append(Problem(
ERROR, "vacuole_link_distance",
f"vacuole_link_distance={distance:g} must be positive.",
"Give the maximum within-vacuole centroid distance in pixels, "
"or leave it blank to derive it from parasite diameter."))
fraction = _numeric(settings.get("non_power_of_two_warn"))
if fraction is not None and not 0 <= fraction <= 1:
problems.append(Problem(
ERROR, "non_power_of_two_warn",
f"non_power_of_two_warn={fraction:g} is not a fraction.",
"Use a value between 0 and 1; 0.2 flags wells above 20%."))
minimum_area = _numeric(settings.get("min_parasite_area"))
maximum_area = _numeric(settings.get("max_parasite_area"))
if (minimum_area is not None and maximum_area is not None
and maximum_area < minimum_area):
problems.append(Problem(
ERROR, "max_parasite_area",
f"max_parasite_area={maximum_area:g} is below "
f"min_parasite_area={minimum_area:g}.",
"Raise the maximum or lower the minimum so at least one "
"parasite size can pass the filter."))
return problems
[docs]
def validate_settings(settings: Dict[str, Any], app_key: str) -> List[Problem]:
"""Check a settings dict against the data it points at.
Image data are checked through headers and directory listings. An enabled
PSF also loads its size-limited kernel to validate calibration and values;
no image processing or GPU inference is run.
:param settings: the settings dict about to be handed to a pipeline.
:param app_key: which pipeline, e.g. ``'mask'``, ``'measure'``,
``'classify'``, ``'umap'``, ``'map_barcodes'``. The
``settings_type`` strings used by the GUI all work, as do a few
aliases (``'sequencing'``, ``'measure_crop'``, ...).
:returns: list of :class:`Problem`; empty means nothing was found.
"""
if not isinstance(settings, dict):
return [Problem(ERROR, "", f"settings must be a dict, got {type(settings).__name__}.",
"Pass the settings dictionary you would hand to the pipeline.")]
app = _normalize_app(app_key)
problems: List[Problem] = []
try:
from .plugins import get_app as _get_plugin_app
known_plugin_app = _get_plugin_app(app) is not None
except Exception:
known_plugin_app = False
if app and app not in APP_FUNCTIONS and not known_plugin_app:
problems.append(Problem(
WARNING, "", f"unknown app '{app_key}'; only the generic checks were run.",
f"Use one of: {', '.join(sorted(APP_FUNCTIONS))}."))
inventories = [_inventory(src, settings, app) for src in _src_values(settings, app)]
problems.extend(_check_src(settings, app, inventories))
problems.extend(_check_channels(settings, app, inventories))
problems.extend(_check_types(settings, app))
problems.extend(_check_unknown_keys(settings, app))
problems.extend(_check_retired_keys(settings))
problems.extend(_check_numeric_sanity(settings))
problems.extend(_check_required_paths(settings, app))
problems.extend(_check_app_specific(settings, app))
try:
from .plugins import get_app, load_object
plugin_app = get_app(app)
if plugin_app is not None and plugin_app.validator:
validator = load_object(plugin_app.validator)
if not callable(validator):
raise TypeError(
f"Plugin validator {plugin_app.validator!r} is not callable"
)
additions = validator(dict(settings))
if additions is None:
additions = ()
if isinstance(additions, Problem):
additions = (additions,)
for item in additions:
if isinstance(item, Problem):
problems.append(item)
elif isinstance(item, dict):
problems.append(Problem(**item))
else:
raise TypeError(
"plugin validator results must be Problem objects or mappings"
)
except Exception as exc:
problems.append(Problem(
ERROR,
"",
f"Plugin validation for '{app or app_key}' failed: {exc}",
"Fix or disable the plugin before running; the pipeline was not started.",
))
return problems
def _render(problem: Problem) -> List[str]:
"""One problem as report lines: the message, then its fix."""
head = f" {problem.setting}: {problem.message}" if problem.setting else f" {problem.message}"
return [head, f" fix: {problem.fix}", ""]
def _fmt(value: Any) -> str:
"""Compact display form for a settings value."""
if value is None:
return "not set"
if isinstance(value, (list, tuple)):
return "[" + ", ".join(str(v) for v in value) + "]"
return str(value)
[docs]
def describe_plan(settings: Dict[str, Any], app_key: str = "") -> str:
"""Summarise what the run would actually do, without doing it.
Reports the app and the function behind it, the resolved source folder,
how many files were found, which objects would be segmented or measured
with which channels and diameters, where output would land, and roughly
how many images would be processed.
:param settings: the settings dict about to be handed to a pipeline.
:param app_key: which pipeline, as for :func:`validate_settings`.
:returns: the plan as a single string, no trailing newline.
"""
if not isinstance(settings, dict):
return "Plan unavailable: settings is not a dict."
app = _normalize_app(app_key)
if app == "regression":
return _describe_regression_plan(settings)
srcs = _src_values(settings, app)
inventories = [_inventory(src, settings, app) for src in srcs]
rows: List[Tuple[str, str]] = []
rows.append(("app", f"{app or 'unknown'}"
+ (f" ({APP_FUNCTIONS[app]})" if app in APP_FUNCTIONS else "")))
named = [str(s) for s in srcs if s not in (None, "")]
rows.append(("source", ", ".join(named) if named else "not set"))
inv = inventories[0] if inventories else _Inventory()
if app in MERGED_APPS and inv.merged_dir and inv.merged_dir != inv.src:
rows.append(("reads", inv.merged_dir))
rows.append(("inputs found", _describe_inputs(inventories)))
if inv.raw_channels is not None:
detail = f"{inv.raw_channels}"
if inv.raw_channel_ids:
detail += f" ({', '.join(inv.raw_channel_ids)})"
rows.append(("channels", detail))
if inv.array_planes is not None:
rows.append(("array planes", f"{inv.array_planes} (indices 0-{inv.array_planes - 1})"))
object_lines = _describe_objects(settings, app)
if object_lines:
label = "would segment" if app == "mask" else "would measure"
rows.append((label, object_lines[0]))
for extra in object_lines[1:]:
rows.append(("", extra))
if app == "measure":
crop_mode = settings.get("crop_mode")
if settings.get("save_png") and isinstance(crop_mode, (list, tuple)) and crop_mode:
size = _fmt(settings.get("png_size"))
rows.append(("crops", f"{', '.join(str(m) for m in crop_mode)} PNGs at {size}"))
rows.append(("measures channels", _fmt(settings.get("channels"))))
for label, path in _describe_outputs(settings, app, inv):
rows.append((label, path))
workload = _describe_workload(settings, app, inventories)
if workload:
rows.append(("workload", workload))
if settings.get("test_mode"):
rows.append(("test mode", "on — only a small subset of the plate would be processed"))
width = max((len(label) for label, _ in rows if label), default=0)
lines = ["Plan — what this run would do. Nothing has been written and no model has been loaded:", ""]
for label, value in rows:
lines.append(f" {label.ljust(width)} {value}" if label else f" {' ' * width} {value}")
return "\n".join(lines)
def _describe_regression_plan(settings: Dict[str, Any]) -> str:
"""Describe paired table inputs without scanning an optional output root.
This is a plan, not a table-content audit. Blank pair members remain
visible; plate and well consistency is checked by the actual loader.
Legacy score/count lists keep their positional pairing, including blanks.
"""
from itertools import zip_longest
rows = [("app", "regression (spacr.ml.perform_regression)")]
pairs = settings.get('paired_data') or []
if not pairs:
def paths(value):
"""Normalize an optional scalar or sequence source into a list without inventing entries."""
return list(value) if isinstance(value, (list, tuple)) else ([] if value is None else [value])
pairs = [dict(score=score, count=count)
for score, count in zip_longest(paths(settings.get('score_data')),
paths(settings.get('count_data')))]
if pairs:
rows.append(('pairing', 'legacy score_data/count_data lists paired by position'))
if not isinstance(pairs, (list, tuple)):
rows.append(('inputs', 'invalid paired_data: expected score/count rows'))
elif not pairs:
rows.append(('inputs', 'no paired_data score/count inputs configured'))
else:
rows.append(('inputs', f'{len(pairs)} paired score/count row(s)'))
for number, pair in enumerate(pairs, 1):
if not isinstance(pair, dict):
rows.append((f'pair {number}', 'invalid row: expected a mapping'))
continue
rows.append((f'pair {number} score', _fmt(pair.get('score') or pair.get('score_data'))))
rows.append((f'pair {number} count', _fmt(pair.get('count') or pair.get('count_data'))))
if pair.get('plate') or pair.get('plateID'):
rows.append((f'pair {number} plate', str(pair.get('plate') or pair.get('plateID'))))
if pair.get('database') or pair.get('measurements'):
rows.append((f'pair {number} linked database', str(pair.get('database') or pair.get('measurements'))))
rows.append(('score column', _fmt(settings.get('dependent_variable'))))
rows.append(('output root', str(settings.get('src') or 'resolved from the paired inputs at run time')))
rows.append(('would write', 'results tables and configured regression plots/reports'))
width = max(len(label) for label, _ in rows)
return '\n'.join([
'Plan — what this run would do. Nothing has been written and no model has been loaded:',
'', *(f' {label.ljust(width)} {value}' for label, value in rows)])
def _describe_inputs(inventories: Sequence[_Inventory]) -> str:
"""One line describing what was found across every ``src``."""
parts: List[str] = []
merged = sum(inv.merged_files for inv in inventories)
stacks = sum(inv.stack_files for inv in inventories)
raw = sum(inv.raw_files for inv in inventories)
missing = [inv.src for inv in inventories if inv.src and not inv.exists]
if missing:
return f"nothing — {', '.join(missing)} does not exist"
if merged:
parts.append(f"{merged} merged array{'' if merged == 1 else 's'} (.npy)")
if stacks:
parts.append(f"{stacks} channel stack{'' if stacks == 1 else 's'} (.npy)")
if raw:
parts.append(f"{raw} raw image file{'' if raw == 1 else 's'}")
return ", ".join(parts) if parts else "no image files found"
def _describe_objects(settings: Dict[str, Any], app: str) -> List[str]:
"""One line per object type the run would work on."""
lines: List[str] = []
for name in OBJECT_NAMES:
if app == "mask":
channel = settings.get(f"{name}_channel")
if channel is None:
continue
diameter = settings.get(f"{name}_diameter")
detail = f"{name}: channel {channel}"
if diameter is not None:
detail += f", diameter {diameter} px"
else:
detail += f", diameter from magnification {settings.get('magnification', 20)}x"
model = settings.get("pathogen_model") if name == "pathogen" else None
if model:
detail += f", model {model}"
if name in SEGMENTED_ROLES[3:]:
detail += f", method {settings.get(f'{name}_method', 'otsu')}"
lines.append(detail)
else:
dim = settings.get(f"{name}_mask_dim")
if dim is None:
continue
detail = f"{name}: mask plane {dim}"
min_size = settings.get(f"{name}_min_size")
if min_size is not None:
detail += f", min size {min_size} px"
lines.append(detail)
if app == "measure" and settings.get("cytoplasm"):
lines.append("cytoplasm: derived from the cell mask")
return lines
def _describe_outputs(settings: Dict[str, Any], app: str, inv: _Inventory) -> List[Tuple[str, str]]:
"""Where the run would write, as ``(label, path)`` rows."""
src = inv.src or str(settings.get("src", ""))
if not src:
return []
rows: List[Tuple[str, str]] = []
if app == "mask":
rows.append(("would write", os.path.join(src, "masks") + os.sep))
rows.append(("", os.path.join(src, "merged") + os.sep))
rows.append(("", os.path.join(src, "settings", "gen_mask_settings.csv")))
elif app == "measure":
root = os.path.dirname(os.path.normpath(inv.merged_dir or src))
rows.append(("would write", os.path.join(root, "measurements", "measurements.db")))
crop_mode = settings.get("crop_mode")
if settings.get("save_png") and isinstance(crop_mode, (list, tuple)):
for mode in crop_mode:
rows.append(("", os.path.join(root, "data", "...", f"{mode}_png") + os.sep))
elif app in DB_APPS:
rows.append(("would read", inv.db_path or os.path.join(src, "measurements", "measurements.db")))
return rows
def _describe_workload(settings: Dict[str, Any], app: str,
inventories: Sequence[_Inventory]) -> str:
"""Rough count of the images the run would touch."""
parts: List[str] = []
if app in MERGED_APPS:
total = sum(inv.merged_files for inv in inventories)
if total:
parts.append(f"~{total} field{'' if total == 1 else 's'} to measure")
elif app == "mask":
fields = sum(inv.fields or 0 for inv in inventories)
stacks = sum(inv.stack_files for inv in inventories)
raw = sum(inv.raw_files for inv in inventories)
if fields:
parts.append(f"~{fields} field{'' if fields == 1 else 's'} from {raw} files")
elif stacks:
parts.append(f"~{stacks} field stack{'' if stacks == 1 else 's'}")
elif raw:
parts.append(f"~{raw} image file{'' if raw == 1 else 's'}")
objects = len([k for k in CHANNEL_KEYS if settings.get(k) is not None])
if objects:
parts.append(f"{objects} segmentation pass{'' if objects == 1 else 'es'} per field")
n_jobs = settings.get("n_jobs")
if n_jobs is not None:
parts.append(f"n_jobs={n_jobs}")
return ", ".join(parts)
#: Directory entries stat'ed when sizing one folder. A plate can hold a
#: million PNG crops; the card must not read them all to say the merged
#: arrays are 60 GB.
_SIZE_BUDGET = 20000
#: Bytes held back from MemAvailable before dividing by the per-worker
#: figure -- the interpreter, the GUI and the page cache all want some.
#: Same reserve :mod:`spacr.benchmark` uses, and for the same reason.
_MEM_RESERVE = 2 * 1024 ** 3
def _fmt_bytes(n: Optional[float]) -> str:
"""``1.4 GB`` for a byte count, ``unknown`` for ``None``."""
if n is None:
return "unknown"
value = float(n)
for unit in ("B", "KB", "MB", "GB"):
if value < 1024:
return f"{value:.0f} {unit}" if unit == "B" else f"{value:.1f} {unit}"
value /= 1024
return f"{value:.1f} TB"
def _dir_bytes(directory: Optional[str],
suffixes: Tuple[str, ...] = ()) -> Tuple[int, int, bool]:
"""Total size of the matching files directly in ``directory``.
Not recursive: every folder this is asked about holds one file per field
and no subtree. Recursing would walk into ``data/`` and its million
crops for no gain.
:returns: ``(total_bytes, files_counted, truncated)``. ``truncated`` says
the budget was hit, so the total is a FLOOR -- reported as ``at
least``, never as the answer.
"""
names = _listdir(directory)
if suffixes:
names = [n for n in names if n.lower().endswith(suffixes)]
truncated = len(names) > _SIZE_BUDGET
total = 0
counted = 0
for name in sorted(names)[:_SIZE_BUDGET]:
try:
total += os.path.getsize(os.path.join(str(directory), name))
counted += 1
except OSError:
continue
return total, counted, truncated
def _array_footprint(directory: Optional[str]) -> Optional[Tuple[Tuple[int, ...], str, int]]:
"""``(shape, dtype, bytes)`` of one ``.npy`` in ``directory``, or None.
Memory-mapped, so the header is read and the pixels are not. This is the
single most useful number on the card: it is what ONE worker must hold,
and it is measured rather than assumed.
"""
names = sorted(f for f in _listdir(directory) if f.endswith(".npy"))
if not names:
return None
import numpy as np
for name in names[:3]:
try:
arr = np.load(os.path.join(str(directory), name), mmap_mode="r")
except (OSError, ValueError, EOFError):
continue
shape = tuple(int(x) for x in getattr(arr, "shape", ()))
if not shape:
continue
dtype = arr.dtype
return shape, str(dtype), int(np.prod(shape) * dtype.itemsize)
return None
def _free_disk(path: Optional[str]) -> Optional[int]:
"""Free bytes on the filesystem holding ``path``, walking up if needed."""
candidate = os.path.abspath(str(path or ""))
for _ in range(6):
try:
return int(shutil.disk_usage(candidate).free)
except OSError:
parent = os.path.dirname(candidate)
if parent == candidate:
return None
candidate = parent
return None
def _gpu_line() -> Optional[str]:
"""One line naming the CUDA device and its free memory, or None.
READ OUT OF ``sys.modules``, NEVER IMPORTED. This module's reason to
exist is that it stays light enough to run before anything heavy is
loaded, and ``test_validate_does_not_import_torch_or_cellpose`` enforces
that -- a lazy import inside this function would still be an import, and
would still cost seconds on a cluster node that only wanted its settings
checked. So the card reports the device when the process already has
torch, and says plainly that it did not look when it does not. Silence
would read as "no GPU", which is a different answer.
"""
torch = sys.modules.get("torch")
if torch is None:
return ("not checked — torch is not loaded in this process, and "
"importing it to ask would cost more than the answer")
try:
if not torch.cuda.is_available():
return "no CUDA device — segmentation would run on the CPU"
index = torch.cuda.current_device()
name = torch.cuda.get_device_name(index)
free, total = torch.cuda.mem_get_info(index)
except Exception:
return None
return (f"{name}, {_fmt_bytes(free)} free of {_fmt_bytes(total)}. "
f"Cellpose tiles, so VRAM follows batch_size and not field size")
[docs]
def describe_resources(settings: Dict[str, Any], app_key: str = "") -> str:
"""Project what the run would COST, without running it.
The second half of the dry-run card. :func:`describe_plan` says what
would happen; this says whether this machine can see it through, which
is the question that actually stops a run at three in the morning.
:param settings: the settings dict about to be handed to a pipeline.
:param app_key: which pipeline, as for :func:`validate_settings`.
:returns: the card as a single string, no trailing newline.
WHAT IS MEASURED, NOT GUESSED: the bytes already on disk, the shape and
dtype of one real array, the free memory, and the free space on the
volume that would be written to. Every one is read off the machine.
WHAT IS DERIVED, AND SAID TO BE: memory is reported as a FLOOR --
``n_jobs`` workers each holding one field is the least the run can use,
before a single working copy. A floor is defensible and still catches
the case that matters, which is eight workers holding a 900 MB field
each on a 16 GB laptop. The measured peak is what
:func:`spacr.benchmark.benchmark` reports, and this card says so rather
than inventing a multiplier for it.
WHAT IS REFUSED: the size of the PNG crop tree. It is objects times crop
modes, and the object count is the thing the run exists to discover. The
card names it as unbounded rather than producing a number that would be
wrong by an order of magnitude either way.
"""
if not isinstance(settings, dict):
return "Resources unavailable: settings is not a dict."
app = _normalize_app(app_key)
srcs = _src_values(settings, app)
inventories = [_inventory(src, settings, app) for src in srcs]
inv = inventories[0] if inventories else _Inventory()
rows: List[Tuple[str, str]] = []
notes: List[str] = []
read_bytes = 0
read_floor = False
for one in inventories:
for directory, suffixes in (
(one.merged_dir if one.merged_exists else None, (".npy",)),
(one.stack_dir, (".npy",)),
(one.raw_dir, IMAGE_EXTENSIONS),
):
if not directory:
continue
total, _, truncated = _dir_bytes(directory, suffixes)
read_bytes += total
read_floor = read_floor or truncated
if read_bytes:
rows.append(("would read",
("at least " if read_floor else "") + _fmt_bytes(read_bytes)))
footprint = _array_footprint(
inv.merged_dir if inv.merged_exists else inv.stack_dir)
per_field: Optional[int] = None
if footprint:
shape, dtype, nbytes = footprint
per_field = nbytes
rows.append(("one field",
f"{'×'.join(str(x) for x in shape)} {dtype} "
f"= {_fmt_bytes(nbytes)} in memory"))
try:
from .benchmark import available_memory_bytes
available = available_memory_bytes()
except Exception:
available = 0
if available:
rows.append(("memory free", _fmt_bytes(available)))
n_jobs = settings.get("n_jobs")
workers = _as_int(n_jobs)
if per_field and workers and workers > 0:
floor = per_field * workers
line = (f"at least {_fmt_bytes(floor)} — {workers} worker(s) × one "
f"field each, before any working copy")
rows.append(("memory needed", line))
if available and floor > max(0, available - _MEM_RESERVE):
safe = max(1, int((available - _MEM_RESERVE) // per_field))
notes.append(
f"n_jobs={workers} does not fit: the floor alone exceeds free "
f"memory less a {_fmt_bytes(_MEM_RESERVE)} reserve. "
f"{safe} worker(s) would fit the floor; run "
f"spacr.benchmark to size it against the real peak.")
else:
notes.append(
"This is a floor, not a peak. spacr.benchmark measures the "
"real per-worker figure on this machine.")
projected: Optional[int] = None
fields = sum(one.merged_files or one.fields or 0 for one in inventories)
if app == "mask" and per_field and fields:
n_objects = len([k for k in CHANNEL_KEYS if settings.get(k) is not None])
merged = per_field * fields
masks = 0
if footprint and len(footprint[0]) >= 2:
plane = footprint[0][0] * footprint[0][1] * 2
masks = plane * fields * max(1, n_objects)
projected = merged + masks
rows.append(("would write",
f"~{_fmt_bytes(projected)} — merged arrays "
f"{_fmt_bytes(merged)} + {max(1, n_objects)} mask "
f"stack(s) {_fmt_bytes(masks)}"))
notes.append(
"merged/ is uncompressed .npy, so it can be LARGER than a "
"compressed TIFF source rather than the same size.")
elif app == "measure":
crop_mode = settings.get("crop_mode")
if settings.get("save_png") and isinstance(crop_mode, (list, tuple)) and crop_mode:
notes.append(
"The PNG crop tree cannot be projected: it is objects × "
f"{len(crop_mode)} crop mode(s), and the object count is "
"what this run exists to discover. Watch the free space "
"above rather than trusting a total.")
try:
from .intensity_rescale import build_plate_plan
for one in inventories:
directory = one.merged_dir if one.merged_exists else one.stack_dir
if not directory:
continue
names = [name for name in _listdir(directory)
if name.endswith('.npy')]
plan = build_plate_plan(directory, names, settings)
scaled = [record for record in plan['fields'].values()
if record['rescale_scope'] == 'plate']
if scaled:
largest = max(float(record['plate_intensity_max'])
for record in scaled)
factors = sorted({float(record['rescale_factor'])
for record in scaled})
notes.append(
"INTENSITY PREFLIGHT: merged values exceed the uint16 "
f"ceiling 65535 (largest plate maximum {largest:g}). "
f"The run will use plate-wide factor(s) "
f"{', '.join(f'{value:g}' for value in factors)} and "
"record one row per field in "
"measurements.db:intensity_rescale.")
if plan['failures']:
notes.append(
"INTENSITY PREFLIGHT: could not inspect "
f"{len(plan['failures'])} field(s); those fields will "
"be explicitly marked as per-field fallbacks if they "
"can be measured.")
except Exception as exc:
notes.append(
"INTENSITY PREFLIGHT: the plate-wide scan could not be "
f"completed ({exc}); workers will record any per-field "
"fallback rather than silently mixing scales.")
write_root = inv.src or (str(srcs[0]) if srcs else "")
free = _free_disk(write_root) if write_root else None
if free is not None:
rows.append(("disk free", f"{_fmt_bytes(free)} on {write_root}"))
if projected is not None and projected > free:
notes.append(
f"NOT ENOUGH DISK: the projection above is "
f"{_fmt_bytes(projected)} and {_fmt_bytes(free)} is free.")
gpu = _gpu_line()
if gpu and app == "mask":
rows.append(("gpu", gpu))
if not read_bytes and footprint is None:
return ("Resources — nothing to project: no readable input was found "
"for this source.")
width = max((len(label) for label, _ in rows if label), default=0)
lines = ["Resources — what this run would cost. Measured where it could "
"be, named where it could not:", ""]
for label, value in rows:
lines.append(f" {label.ljust(width)} {value}" if label
else f" {' ' * width} {value}")
for note in notes:
lines.append("")
lines.append(f" {note}")
return "\n".join(lines)
DRY_RUN_TRAILER = "dry_run=True — stopping here. Set dry_run=False to run for real."
[docs]
def run_preflight(settings: Dict[str, Any], app_key: str, printer=print,
trailer: str = DRY_RUN_TRAILER) -> List[Problem]:
"""Validate, print the report and the plan, and hand back the problems.
This is what the ``dry_run`` branch of every pipeline entry point calls,
so the wording is identical wherever it is triggered from.
:param settings: the settings dict that would have been run.
:param app_key: which pipeline, as for :func:`validate_settings`.
:param printer: where the text goes; defaults to ``print``.
:param trailer: the closing line. The default names the in-pipeline
``dry_run`` flag, which is the wrong advice for a caller reached some
other way -- ``spacr-run --dry-run`` passes its own wording. Pass an
empty string to suppress it entirely.
:returns: the list of :class:`Problem` found.
"""
problems = validate_settings(settings, app_key)
printer(format_report(problems, settings, app_key))
printer("")
printer(describe_plan(settings, app_key))
try:
printer("")
printer(describe_resources(settings, app_key))
except Exception as exc:
printer("")
printer(f"Resources — could not be projected: {exc}")
if trailer:
printer("")
printer(trailer)
return problems